LLVM 24.0.0git
AMDGPUAsmParser.cpp
Go to the documentation of this file.
1//===- AMDGPUAsmParser.cpp - Parse SI asm to MCInst instructions ----------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8
9#include "AMDKernelCodeT.h"
16#include "SIDefines.h"
17#include "SIInstrInfo.h"
22#include "llvm/ADT/APFloat.h"
24#include "llvm/ADT/Twine.h"
27#include "llvm/MC/MCAsmInfo.h"
28#include "llvm/MC/MCContext.h"
29#include "llvm/MC/MCExpr.h"
30#include "llvm/MC/MCInst.h"
31#include "llvm/MC/MCInstrDesc.h"
37#include "llvm/MC/MCSymbol.h"
46#include <optional>
47
48using namespace llvm;
49using namespace llvm::AMDGPU;
50using namespace llvm::amdhsa;
51
52namespace {
53
54class AMDGPUAsmParser;
55
56enum RegisterKind {
57 IS_UNKNOWN,
58 IS_VGPR,
59 IS_SGPR,
60 IS_AGPR,
61 IS_TTMP,
62 IS_SPECIAL
63};
64
65//===----------------------------------------------------------------------===//
66// Operand
67//===----------------------------------------------------------------------===//
68
69class AMDGPUOperand : public MCParsedAsmOperand {
70 enum KindTy { Token, Immediate, Register, Expression } Kind;
71
72 SMLoc StartLoc, EndLoc;
73 const AMDGPUAsmParser *AsmParser;
74
75public:
76 AMDGPUOperand(KindTy Kind_, const AMDGPUAsmParser *AsmParser_)
77 : Kind(Kind_), AsmParser(AsmParser_) {}
78
79 using Ptr = std::unique_ptr<AMDGPUOperand>;
80
81 struct Modifiers {
82 bool Abs = false;
83 bool Neg = false;
84 bool Sext = false;
85 LitModifier Lit = LitModifier::None;
86
87 bool hasFPModifiers() const { return Abs || Neg; }
88 bool hasIntModifiers() const { return Sext; }
89 bool hasModifiers() const { return hasFPModifiers() || hasIntModifiers(); }
90 bool isForcedLit() const { return Lit == LitModifier::Lit; }
91 bool isForcedLit64() const { return Lit == LitModifier::Lit64; }
92
93 int64_t getFPModifiersOperand() const {
94 int64_t Operand = 0;
95 Operand |= Abs ? SISrcMods::ABS : 0u;
96 Operand |= Neg ? SISrcMods::NEG : 0u;
97 return Operand;
98 }
99
100 int64_t getIntModifiersOperand() const {
101 int64_t Operand = 0;
102 Operand |= Sext ? SISrcMods::SEXT : 0u;
103 return Operand;
104 }
105
106 int64_t getModifiersOperand() const {
107 assert(!(hasFPModifiers() && hasIntModifiers()) &&
108 "fp and int modifiers should not be used simultaneously");
109 if (hasFPModifiers())
110 return getFPModifiersOperand();
111 if (hasIntModifiers())
112 return getIntModifiersOperand();
113 return 0;
114 }
115
116 friend raw_ostream &operator<<(raw_ostream &OS,
117 AMDGPUOperand::Modifiers Mods);
118 };
119
120 enum ImmTy {
121 ImmTyNone,
122 ImmTyGDS,
123 ImmTyLDS,
124 ImmTyOffen,
125 ImmTyIdxen,
126 ImmTyAddr64,
127 ImmTyOffset,
128 ImmTyInstOffset,
129 ImmTyOffset0,
130 ImmTyOffset1,
131 ImmTySMEMOffsetMod,
132 ImmTyCPol,
133 ImmTyTFE,
134 ImmTyIsAsync,
135 ImmTyD16,
136 ImmTyClamp,
137 ImmTyOModSI,
138 ImmTySDWADstSel,
139 ImmTySDWASrc0Sel,
140 ImmTySDWASrc1Sel,
141 ImmTySDWADstUnused,
142 ImmTyDMask,
143 ImmTyDim,
144 ImmTyUNorm,
145 ImmTyDA,
146 ImmTyR128A16,
147 ImmTyA16,
148 ImmTyLWE,
149 ImmTyExpTgt,
150 ImmTyExpCompr,
151 ImmTyExpVM,
152 ImmTyDone,
153 ImmTyRowEn,
154 ImmTyFORMAT,
155 ImmTyHwreg,
156 ImmTyOff,
157 ImmTySendMsg,
158 ImmTyWaitEvent,
159 ImmTyInterpSlot,
160 ImmTyInterpAttr,
161 ImmTyInterpAttrChan,
162 ImmTyOpSel,
163 ImmTyOpSelHi,
164 ImmTyNegLo,
165 ImmTyNegHi,
166 ImmTyIndexKey8bit,
167 ImmTyIndexKey16bit,
168 ImmTyIndexKey32bit,
169 ImmTyDPP8,
170 ImmTyDppCtrl,
171 ImmTyDppRowMask,
172 ImmTyDppBankMask,
173 ImmTyDppBoundCtrl,
174 ImmTyDppFI,
175 ImmTySwizzle,
176 ImmTyGprIdxMode,
177 ImmTyHigh,
178 ImmTyBLGP,
179 ImmTyCBSZ,
180 ImmTyABID,
181 ImmTyEndpgm,
182 ImmTyWaitVDST,
183 ImmTyWaitEXP,
184 ImmTyWaitVAVDst,
185 ImmTyWaitVMVSrc,
186 ImmTyBitOp3,
187 ImmTyMatrixAFMT,
188 ImmTyMatrixBFMT,
189 ImmTyMatrixAScale,
190 ImmTyMatrixBScale,
191 ImmTyMatrixAScaleFmt,
192 ImmTyMatrixBScaleFmt,
193 ImmTyMatrixAReuse,
194 ImmTyMatrixBReuse,
195 ImmTyScaleSel,
196 ImmTyByteSel,
197 };
198
199private:
200 struct TokOp {
201 const char *Data;
202 unsigned Length;
203 };
204
205 struct ImmOp {
206 int64_t Val;
207 ImmTy Type;
208 bool IsFPImm;
209 Modifiers Mods;
210 };
211
212 struct RegOp {
213 MCRegister RegNo;
214 Modifiers Mods;
215 };
216
217 union {
218 TokOp Tok;
219 ImmOp Imm;
220 RegOp Reg;
221 const MCExpr *Expr;
222 };
223
224 // The index of the associated MCInst operand.
225 mutable int MCOpIdx = -1;
226
227public:
228 bool isToken() const override { return Kind == Token; }
229
230 bool isSymbolRefExpr() const {
231 return isExpr() && Expr && isa<MCSymbolRefExpr>(Expr);
232 }
233
234 bool isImm() const override { return Kind == Immediate; }
235
236 bool isInlinableImm(MVT type) const;
237 bool isLiteralImm(MVT type) const;
238
239 bool isRegKind() const { return Kind == Register; }
240
241 bool isReg() const override { return isRegKind() && !hasModifiers(); }
242
243 bool isRegOrInline(unsigned RCID, MVT type) const {
244 return isRegClass(RCID) || isInlinableImm(type);
245 }
246
247 bool isRegOrInlineTarget(unsigned TargetRCIdx, MVT type) const {
248 return isRegClassTarget(TargetRCIdx) || isInlinableImm(type);
249 }
250
251 bool isRegOrImmWithInputMods(unsigned RCID, MVT type) const {
252 return isRegOrInline(RCID, type) || isLiteralImm(type);
253 }
254
255 bool isRegOrImmWithInputModsTarget(unsigned TargetRCIdx, MVT type) const {
256 return isRegOrInlineTarget(TargetRCIdx, type) || isLiteralImm(type);
257 }
258
259 template <bool IsFake16> bool isRegOrImmWithIntT16InputMods() const {
261 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
262 }
263
264 bool isRegOrImmWithInt32InputMods() const {
265 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::i32);
266 }
267
268 template <bool IsFake16> bool isRegOrInlineImmWithIntT16InputMods() const {
269 return isRegOrInline(
270 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
271 }
272
273 bool isRegOrInlineImmWithInt32InputMods() const {
274 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i32);
275 }
276
277 bool isRegOrImmWithInt64InputMods() const {
278 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
279 }
280
281 bool isRegOrImmWithFP16InputMods() const {
282 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::f16);
283 }
284
285 template <bool IsFake16> bool isRegOrImmWithFPT16InputMods() const {
287 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
288 }
289
290 bool isRegOrImmWithFPT16_LO16InputMods() const {
291 return isRegOrImmWithInputMods(AMDGPU::VS_16_LO16RegClassID, MVT::f16);
292 }
293
294 bool isRegOrImmWithFP32InputMods() const {
295 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::f32);
296 }
297
298 bool isRegOrImmWithFP64InputMods() const {
299 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
300 }
301
302 template <bool IsFake16> bool isRegOrInlineImmWithFP16InputMods() const {
303 return isRegOrInline(
304 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
305 }
306
307 bool isRegOrInlineImmWithFP32InputMods() const {
308 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::f32);
309 }
310
311 bool isRegOrInlineImmWithFP64InputMods() const {
312 return isRegOrInlineTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
313 }
314
315 bool isVRegWithInputMods(unsigned RCID) const { return isRegClass(RCID); }
316
317 bool isVRegWithFP32InputMods() const {
318 return isVRegWithInputMods(AMDGPU::VGPR_32RegClassID);
319 }
320
321 bool isVRegWithFP64InputMods() const {
322 return isRegClassTarget(AMDGPU::VReg_64_AlignTarget);
323 }
324
325 bool isPackedFP16InputMods() const {
326 return isRegOrImmWithInputMods(AMDGPU::VS_32RegClassID, MVT::v2f16);
327 }
328
329 bool isPackedVGPRFP32InputMods() const {
330 return isRegOrImmWithInputMods(AMDGPU::VReg_64RegClassID, MVT::v2f32);
331 }
332
333 bool isVReg32() const { return isRegClass(AMDGPU::VGPR_32RegClassID); }
334
335 bool isVReg32OrOff() const { return isOff() || isVReg32(); }
336
337 bool isRsrcReg32() const { return isRegClass(AMDGPU::RsrcReg32RegClassID); }
338
339 bool isNull() const { return isRegKind() && getReg() == AMDGPU::SGPR_NULL; }
340
341 bool isAV_LdSt_32_Align2_RegOp() const {
342 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
343 isRegClass(AMDGPU::AGPR_32RegClassID);
344 }
345
346 bool isVRegWithInputMods() const;
347 template <bool IsFake16> bool isT16_Lo128VRegWithInputMods() const;
348 template <bool IsFake16> bool isT16VRegWithInputMods() const;
349 bool isT16_LO16VRegWithInputMods() const;
350
351 bool isSDWAOperand(MVT type) const;
352 bool isSDWAFP16Operand() const;
353 bool isSDWAFP32Operand() const;
354 bool isSDWAInt16Operand() const;
355 bool isSDWAInt32Operand() const;
356
357 bool isImmTy(ImmTy ImmT) const { return isImm() && Imm.Type == ImmT; }
358
359 template <ImmTy Ty> bool isImmTy() const { return isImmTy(Ty); }
360
361 bool isImmLiteral() const { return isImmTy(ImmTyNone); }
362
363 bool isImmModifier() const { return isImm() && Imm.Type != ImmTyNone; }
364
365 bool isOModSI() const { return isImmTy(ImmTyOModSI); }
366 bool isDim() const { return isImmTy(ImmTyDim); }
367 bool isR128A16() const { return isImmTy(ImmTyR128A16); }
368 bool isOff() const { return isImmTy(ImmTyOff); }
369 bool isExpTgt() const { return isImmTy(ImmTyExpTgt); }
370 bool isOffen() const { return isImmTy(ImmTyOffen); }
371 bool isIdxen() const { return isImmTy(ImmTyIdxen); }
372 bool isAddr64() const { return isImmTy(ImmTyAddr64); }
373 bool isSMEMOffsetMod() const { return isImmTy(ImmTySMEMOffsetMod); }
374 bool isFlatOffset() const {
375 return isImmTy(ImmTyOffset) || isImmTy(ImmTyInstOffset);
376 }
377 bool isGDS() const { return isImmTy(ImmTyGDS); }
378 bool isLDS() const { return isImmTy(ImmTyLDS); }
379 bool isCPol() const { return isImmTy(ImmTyCPol); }
380 bool isIndexKey8bit() const { return isImmTy(ImmTyIndexKey8bit); }
381 bool isIndexKey16bit() const { return isImmTy(ImmTyIndexKey16bit); }
382 bool isIndexKey32bit() const { return isImmTy(ImmTyIndexKey32bit); }
383 bool isMatrixAFMT() const { return isImmTy(ImmTyMatrixAFMT); }
384 bool isMatrixBFMT() const { return isImmTy(ImmTyMatrixBFMT); }
385 bool isMatrixAScale() const { return isImmTy(ImmTyMatrixAScale); }
386 bool isMatrixBScale() const { return isImmTy(ImmTyMatrixBScale); }
387 bool isMatrixAScaleFmt() const { return isImmTy(ImmTyMatrixAScaleFmt); }
388 bool isMatrixBScaleFmt() const { return isImmTy(ImmTyMatrixBScaleFmt); }
389 bool isTFE() const { return isImmTy(ImmTyTFE); }
390 bool isFORMAT() const { return isImmTy(ImmTyFORMAT) && isUInt<7>(getImm()); }
391 bool isDppFI() const { return isImmTy(ImmTyDppFI); }
392 bool isSDWADstSel() const { return isImmTy(ImmTySDWADstSel); }
393 bool isSDWASrc0Sel() const { return isImmTy(ImmTySDWASrc0Sel); }
394 bool isSDWASrc1Sel() const { return isImmTy(ImmTySDWASrc1Sel); }
395 bool isSDWADstUnused() const { return isImmTy(ImmTySDWADstUnused); }
396 bool isInterpSlot() const { return isImmTy(ImmTyInterpSlot); }
397 bool isInterpAttr() const { return isImmTy(ImmTyInterpAttr); }
398 bool isInterpAttrChan() const { return isImmTy(ImmTyInterpAttrChan); }
399 bool isOpSel() const { return isImmTy(ImmTyOpSel); }
400 bool isOpSelHi() const { return isImmTy(ImmTyOpSelHi); }
401 bool isNegLo() const { return isImmTy(ImmTyNegLo); }
402 bool isNegHi() const { return isImmTy(ImmTyNegHi); }
403 bool isDone() const { return isImmTy(ImmTyDone); }
404 bool isRowEn() const { return isImmTy(ImmTyRowEn); }
405
406 bool isRegOrImm() const { return isReg() || isImm(); }
407
408 bool isRegClass(unsigned RCID) const;
409
410 // Check the register against the HwMode-resolved operand class.
411 bool isRegClassTarget(unsigned TargetRCIdx) const;
412
413 bool isInlineValue() const;
414
415 bool isRegOrInlineNoMods(unsigned RCID, MVT type) const {
416 return isRegOrInline(RCID, type) && !hasModifiers();
417 }
418
419 bool isRegOrInlineNoModsTarget(unsigned TargetRCIdx, MVT type) const {
420 return isRegOrInlineTarget(TargetRCIdx, type) && !hasModifiers();
421 }
422
423 bool isSCSrcB16() const {
424 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i16);
425 }
426
427 bool isSCSrc_b32() const {
428 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i32);
429 }
430
431 bool isSCSrc_b64() const {
432 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::i64);
433 }
434
435 bool isBoolReg() const;
436
437 bool isSSrc_b32() const {
438 return isSCSrc_b32() || isLiteralImm(MVT::i32) || isExpr();
439 }
440
441 bool isSSrc_b16() const { return isSCSrcB16() || isLiteralImm(MVT::i16); }
442
443 bool isSSrc_b64() const {
444 // TODO: Find out how SALU supports extension of 32-bit literals to 64 bits.
445 // See isVSrc64().
446 return isSCSrc_b64() || isLiteralImm(MVT::i64) ||
447 (((const MCTargetAsmParser *)AsmParser)
448 ->getAvailableFeatures()[AMDGPU::Feature64BitLiterals] &&
449 isExpr());
450 }
451
452 bool isSSrc_f32() const {
453 return isSCSrc_b32() || isLiteralImm(MVT::f32) || isExpr();
454 }
455
456 bool isSSrc_bf16() const { return isSCSrcB16() || isLiteralImm(MVT::bf16); }
457
458 bool isSSrc_f16() const { return isSCSrcB16() || isLiteralImm(MVT::f16); }
459
460 bool isSSrc_NoInline_f16() const { return isSSrc_f16(); }
461
462 bool isSSrcOrLds_b32() const {
463 return isRegOrInlineNoMods(AMDGPU::SRegOrLds_32RegClassID, MVT::i32) ||
464 isLiteralImm(MVT::i32) || isExpr();
465 }
466
467 bool isVCSrc_b32() const {
468 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i32);
469 }
470
471 bool isVCSrc_b32_Lo256() const {
472 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo256RegClassID, MVT::i32);
473 }
474
475 bool isVCSrc_b64_Lo256() const {
476 return isRegOrInlineNoMods(AMDGPU::VS_64_Lo256RegClassID, MVT::i64);
477 }
478
479 bool isVCSrc_b64() const {
480 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
481 }
482
483 bool isVCSrcT_b16() const {
484 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::i16);
485 }
486
487 bool isVCSrcTB16_Lo128() const {
488 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::i16);
489 }
490
491 bool isVCSrcFake16B16_Lo128() const {
492 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::i16);
493 }
494
495 bool isVCSrc_b16() const {
496 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i16);
497 }
498
499 bool isVCSrc_v2b16() const { return isVCSrc_b16(); }
500
501 bool isVCSrc_f32() const {
502 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f32);
503 }
504
505 bool isVCSrc_f64() const {
506 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
507 }
508
509 bool isVCSrcTBF16() const {
510 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::bf16);
511 }
512
513 bool isVCSrcT_f16() const {
514 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
515 }
516
517 bool isVCSrcT_bf16() const {
518 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
519 }
520
521 bool isVCSrcTF16_LO16() const {
522 return isRegOrInlineNoMods(AMDGPU::VS_16_LO16RegClassID, MVT::f16);
523 }
524
525 bool isVCSrcTBF16_Lo128() const {
526 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::bf16);
527 }
528
529 bool isVCSrcTF16_Lo128() const {
530 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::f16);
531 }
532
533 bool isVCSrcFake16BF16_Lo128() const {
534 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::bf16);
535 }
536
537 bool isVCSrcFake16F16_Lo128() const {
538 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::f16);
539 }
540
541 bool isVCSrc_bf16() const {
542 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::bf16);
543 }
544
545 bool isVCSrc_f16() const {
546 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f16);
547 }
548
549 bool isVCSrc_v2bf16() const { return isVCSrc_bf16(); }
550
551 bool isVCSrc_v2f16() const { return isVCSrc_f16(); }
552
553 bool isVSrc_b32() const {
554 return isVCSrc_f32() || isLiteralImm(MVT::i32) || isExpr();
555 }
556
557 bool isVSrc_b64() const { return isVCSrc_f64() || isLiteralImm(MVT::i64); }
558
559 bool isVSrc_v2b64() const {
560 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::i64) ||
561 isLiteralImm(MVT::i64);
562 }
563
564 bool isVSrc_v2f64() const {
565 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::f64) ||
566 isLiteralImm(MVT::f64);
567 }
568
569 bool isVSrcT_b16() const { return isVCSrcT_b16() || isLiteralImm(MVT::i16); }
570
571 bool isVSrcT_b16_Lo128() const {
572 return isVCSrcTB16_Lo128() || isLiteralImm(MVT::i16);
573 }
574
575 bool isVSrcFake16_b16_Lo128() const {
576 return isVCSrcFake16B16_Lo128() || isLiteralImm(MVT::i16);
577 }
578
579 bool isVSrc_b16() const { return isVCSrc_b16() || isLiteralImm(MVT::i16); }
580
581 bool isVSrc_v2b16() const { return isVSrc_b16() || isLiteralImm(MVT::v2i16); }
582
583 bool isVSrc_v2f32() const { return isVSrc_f64() || isLiteralImm(MVT::v2f32); }
584
585 bool isVCSrc_v2b32() const { return isVCSrc_b64(); }
586
587 bool isVSrc_v2b32() const { return isVSrc_b64() || isLiteralImm(MVT::v2i32); }
588
589 bool isVSrc_f32() const {
590 return isVCSrc_f32() || isLiteralImm(MVT::f32) || isExpr();
591 }
592
593 bool isVSrc_f64() const {
594 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64) ||
595 isLiteralImm(MVT::f64);
596 }
597
598 bool isVSrcT_bf16() const {
599 return isVCSrcTBF16() || isLiteralImm(MVT::bf16);
600 }
601
602 bool isVSrcT_f16() const { return isVCSrcT_f16() || isLiteralImm(MVT::f16); }
603
604 bool isVSrcT_f16_LO16() const {
605 return isVCSrcTF16_LO16() || isLiteralImm(MVT::f16);
606 }
607
608 bool isVSrcT_bf16_Lo128() const {
609 return isVCSrcTBF16_Lo128() || isLiteralImm(MVT::bf16);
610 }
611
612 bool isVSrcT_f16_Lo128() const {
613 return isVCSrcTF16_Lo128() || isLiteralImm(MVT::f16);
614 }
615
616 bool isVSrcFake16_bf16_Lo128() const {
617 return isVCSrcFake16BF16_Lo128() || isLiteralImm(MVT::bf16);
618 }
619
620 bool isVSrcFake16_f16_Lo128() const {
621 return isVCSrcFake16F16_Lo128() || isLiteralImm(MVT::f16);
622 }
623
624 bool isVSrc_bf16() const { return isVCSrc_bf16() || isLiteralImm(MVT::bf16); }
625
626 bool isVSrc_f16() const { return isVCSrc_f16() || isLiteralImm(MVT::f16); }
627
628 bool isVSrc_v2bf16() const {
629 return isVSrc_bf16() || isLiteralImm(MVT::v2bf16);
630 }
631
632 bool isVSrc_v2f16() const { return isVSrc_f16() || isLiteralImm(MVT::v2f16); }
633
634 bool isVSrc_v2f16_splat() const { return isVSrc_v2f16(); }
635
636 bool isVSrc_NoInline_v2f16() const { return isVSrc_v2f16(); }
637
638 bool isVISrc_64_bf16() const {
639 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::bf16);
640 }
641
642 bool isVISrc_64_f16() const {
643 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f16);
644 }
645
646 bool isVISrc_64_b32() const {
647 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
648 }
649
650 bool isVISrc_64_f64() const {
651 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f64);
652 }
653
654 bool isVISrc_256_b32() const {
655 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
656 }
657
658 bool isVISrc_256_f32() const {
659 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
660 }
661
662 bool isVISrc_256_f64() const {
663 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f64);
664 }
665
666 bool isVISrc_512_f64() const {
667 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f64);
668 }
669
670 bool isVISrc_128_b32() const {
671 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i32);
672 }
673
674 bool isVISrc_128_f32() const {
675 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f32);
676 }
677
678 bool isVISrc_512_b32() const {
679 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i32);
680 }
681
682 bool isVISrc_512_f32() const {
683 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f32);
684 }
685
686 bool isVISrc_1024_b32() const {
687 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i32);
688 }
689
690 bool isVISrc_1024_f32() const {
691 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f32);
692 }
693
694 bool isAISrc_64_f64() const {
695 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::f64);
696 }
697
698 bool isAISrc_128_b32() const {
699 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i32);
700 }
701
702 bool isAISrc_128_f32() const {
703 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f32);
704 }
705
706 bool isVISrc_128_bf16() const {
707 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::bf16);
708 }
709
710 bool isVISrc_128_f16() const {
711 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f16);
712 }
713
714 bool isAISrc_256_f64() const {
715 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::f64);
716 }
717
718 bool isAISrc_512_b32() const {
719 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i32);
720 }
721
722 bool isAISrc_512_f32() const {
723 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f32);
724 }
725
726 bool isAISrc_1024_b32() const {
727 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i32);
728 }
729
730 bool isAISrc_1024_f32() const {
731 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f32);
732 }
733
734 bool isKImmFP32() const { return isLiteralImm(MVT::f32); }
735
736 bool isKImmFP16() const { return isLiteralImm(MVT::f16); }
737
738 bool isKImmFP64() const { return isLiteralImm(MVT::f64); }
739
740 bool isMem() const override { return false; }
741
742 bool isExpr() const { return Kind == Expression; }
743
744 bool isSOPPBrTarget() const { return isExpr() || isImm(); }
745
746 bool isSWaitCnt() const;
747 bool isDepCtr() const;
748 bool isSDelayALU() const;
749 bool isHwreg() const;
750 bool isSendMsg() const;
751 bool isWaitEvent() const;
752 bool isSplitBarrier() const;
753 bool isSwizzle() const;
754 bool isSMRDOffset8() const;
755 bool isSMEMOffset() const;
756 bool isSMRDLiteralOffset() const;
757 bool isDPP8() const;
758 bool isDPPCtrl() const;
759 bool isBLGP() const;
760 bool isGPRIdxMode() const;
761 bool isS16Imm() const;
762 bool isU16Imm() const;
763 bool isEndpgm() const;
764
765 auto getPredicate(std::function<bool(const AMDGPUOperand &Op)> P) const {
766 return [this, P]() { return P(*this); };
767 }
768
769 StringRef getToken() const {
770 assert(isToken());
771 return StringRef(Tok.Data, Tok.Length);
772 }
773
774 int64_t getImm() const {
775 assert(isImm());
776 return Imm.Val;
777 }
778
779 void setImm(int64_t Val) {
780 assert(isImm());
781 Imm.Val = Val;
782 }
783
784 ImmTy getImmTy() const {
785 assert(isImm());
786 return Imm.Type;
787 }
788
789 MCRegister getReg() const override {
790 assert(isRegKind());
791 return Reg.RegNo;
792 }
793
794 SMLoc getStartLoc() const override { return StartLoc; }
795
796 SMLoc getEndLoc() const override { return EndLoc; }
797
798 int getMCOpIdx() const { return MCOpIdx; }
799
800 Modifiers getModifiers() const {
801 assert(isRegKind() || isImmTy(ImmTyNone));
802 return isRegKind() ? Reg.Mods : Imm.Mods;
803 }
804
805 void setModifiers(Modifiers Mods) {
806 assert(isRegKind() || isImmTy(ImmTyNone));
807 if (isRegKind())
808 Reg.Mods = Mods;
809 else
810 Imm.Mods = Mods;
811 }
812
813 bool hasModifiers() const { return getModifiers().hasModifiers(); }
814
815 bool hasFPModifiers() const { return getModifiers().hasFPModifiers(); }
816
817 bool hasIntModifiers() const { return getModifiers().hasIntModifiers(); }
818
819 bool isForcedLit() const {
820 return isImmLiteral() && getModifiers().isForcedLit();
821 }
822
823 bool isForcedLit64() const {
824 return isImmLiteral() && getModifiers().isForcedLit64();
825 }
826
827 uint64_t applyInputFPModifiers(uint64_t Val, unsigned Size) const;
828
829 void addImmOperands(MCInst &Inst, unsigned N,
830 bool ApplyModifiers = true) const;
831
832 void addLiteralImmOperand(MCInst &Inst, int64_t Val,
833 bool ApplyModifiers) const;
834
835 void addRegOperands(MCInst &Inst, unsigned N) const;
836
837 void addRegOrImmOperands(MCInst &Inst, unsigned N) const {
838 if (isRegKind())
839 addRegOperands(Inst, N);
840 else
841 addImmOperands(Inst, N);
842 }
843
844 void addRegOrImmWithInputModsOperands(MCInst &Inst, unsigned N) const {
845 Modifiers Mods = getModifiers();
846 Inst.addOperand(MCOperand::createImm(Mods.getModifiersOperand()));
847 if (isRegKind()) {
848 addRegOperands(Inst, N);
849 } else {
850 addImmOperands(Inst, N, false);
851 }
852 }
853
854 void addRegOrImmWithFPInputModsOperands(MCInst &Inst, unsigned N) const {
855 assert(!hasIntModifiers());
856 addRegOrImmWithInputModsOperands(Inst, N);
857 }
858
859 void addRegWithInputModsOperands(MCInst &Inst, unsigned N) const {
860 Modifiers Mods = getModifiers();
861 Inst.addOperand(MCOperand::createImm(Mods.getModifiersOperand()));
862 assert(isRegKind());
863 addRegOperands(Inst, N);
864 }
865
866 void addRegWithFPInputModsOperands(MCInst &Inst, unsigned N) const {
867 assert(!hasIntModifiers());
868 addRegWithInputModsOperands(Inst, N);
869 }
870
871 static void printImmTy(raw_ostream &OS, ImmTy Type) {
872 // clang-format off
873 switch (Type) {
874 case ImmTyNone: OS << "None"; break;
875 case ImmTyGDS: OS << "GDS"; break;
876 case ImmTyLDS: OS << "LDS"; break;
877 case ImmTyOffen: OS << "Offen"; break;
878 case ImmTyIdxen: OS << "Idxen"; break;
879 case ImmTyAddr64: OS << "Addr64"; break;
880 case ImmTyOffset: OS << "Offset"; break;
881 case ImmTyInstOffset: OS << "InstOffset"; break;
882 case ImmTyOffset0: OS << "Offset0"; break;
883 case ImmTyOffset1: OS << "Offset1"; break;
884 case ImmTySMEMOffsetMod: OS << "SMEMOffsetMod"; break;
885 case ImmTyCPol: OS << "CPol"; break;
886 case ImmTyIndexKey8bit: OS << "index_key"; break;
887 case ImmTyIndexKey16bit: OS << "index_key"; break;
888 case ImmTyIndexKey32bit: OS << "index_key"; break;
889 case ImmTyTFE: OS << "TFE"; break;
890 case ImmTyIsAsync: OS << "IsAsync"; break;
891 case ImmTyD16: OS << "D16"; break;
892 case ImmTyFORMAT: OS << "FORMAT"; break;
893 case ImmTyClamp: OS << "Clamp"; break;
894 case ImmTyOModSI: OS << "OModSI"; break;
895 case ImmTyDPP8: OS << "DPP8"; break;
896 case ImmTyDppCtrl: OS << "DppCtrl"; break;
897 case ImmTyDppRowMask: OS << "DppRowMask"; break;
898 case ImmTyDppBankMask: OS << "DppBankMask"; break;
899 case ImmTyDppBoundCtrl: OS << "DppBoundCtrl"; break;
900 case ImmTyDppFI: OS << "DppFI"; break;
901 case ImmTySDWADstSel: OS << "SDWADstSel"; break;
902 case ImmTySDWASrc0Sel: OS << "SDWASrc0Sel"; break;
903 case ImmTySDWASrc1Sel: OS << "SDWASrc1Sel"; break;
904 case ImmTySDWADstUnused: OS << "SDWADstUnused"; break;
905 case ImmTyDMask: OS << "DMask"; break;
906 case ImmTyDim: OS << "Dim"; break;
907 case ImmTyUNorm: OS << "UNorm"; break;
908 case ImmTyDA: OS << "DA"; break;
909 case ImmTyR128A16: OS << "R128A16"; break;
910 case ImmTyA16: OS << "A16"; break;
911 case ImmTyLWE: OS << "LWE"; break;
912 case ImmTyOff: OS << "Off"; break;
913 case ImmTyExpTgt: OS << "ExpTgt"; break;
914 case ImmTyExpCompr: OS << "ExpCompr"; break;
915 case ImmTyExpVM: OS << "ExpVM"; break;
916 case ImmTyDone: OS << "Done"; break;
917 case ImmTyRowEn: OS << "RowEn"; break;
918 case ImmTyHwreg: OS << "Hwreg"; break;
919 case ImmTySendMsg: OS << "SendMsg"; break;
920 case ImmTyWaitEvent: OS << "WaitEvent"; break;
921 case ImmTyInterpSlot: OS << "InterpSlot"; break;
922 case ImmTyInterpAttr: OS << "InterpAttr"; break;
923 case ImmTyInterpAttrChan: OS << "InterpAttrChan"; break;
924 case ImmTyOpSel: OS << "OpSel"; break;
925 case ImmTyOpSelHi: OS << "OpSelHi"; break;
926 case ImmTyNegLo: OS << "NegLo"; break;
927 case ImmTyNegHi: OS << "NegHi"; break;
928 case ImmTySwizzle: OS << "Swizzle"; break;
929 case ImmTyGprIdxMode: OS << "GprIdxMode"; break;
930 case ImmTyHigh: OS << "High"; break;
931 case ImmTyBLGP: OS << "BLGP"; break;
932 case ImmTyCBSZ: OS << "CBSZ"; break;
933 case ImmTyABID: OS << "ABID"; break;
934 case ImmTyEndpgm: OS << "Endpgm"; break;
935 case ImmTyWaitVDST: OS << "WaitVDST"; break;
936 case ImmTyWaitEXP: OS << "WaitEXP"; break;
937 case ImmTyWaitVAVDst: OS << "WaitVAVDst"; break;
938 case ImmTyWaitVMVSrc: OS << "WaitVMVSrc"; break;
939 case ImmTyBitOp3: OS << "BitOp3"; break;
940 case ImmTyMatrixAFMT: OS << "ImmTyMatrixAFMT"; break;
941 case ImmTyMatrixBFMT: OS << "ImmTyMatrixBFMT"; break;
942 case ImmTyMatrixAScale: OS << "ImmTyMatrixAScale"; break;
943 case ImmTyMatrixBScale: OS << "ImmTyMatrixBScale"; break;
944 case ImmTyMatrixAScaleFmt: OS << "ImmTyMatrixAScaleFmt"; break;
945 case ImmTyMatrixBScaleFmt: OS << "ImmTyMatrixBScaleFmt"; break;
946 case ImmTyMatrixAReuse: OS << "ImmTyMatrixAReuse"; break;
947 case ImmTyMatrixBReuse: OS << "ImmTyMatrixBReuse"; break;
948 case ImmTyScaleSel: OS << "ScaleSel" ; break;
949 case ImmTyByteSel: OS << "ByteSel" ; break;
950 }
951 // clang-format on
952 }
953
954 void print(raw_ostream &OS, const MCAsmInfo &MAI) const override {
955 switch (Kind) {
956 case Register:
957 OS << "<register " << AMDGPUInstPrinter::getRegisterName(getReg())
958 << " mods: " << Reg.Mods << '>';
959 break;
960 case Immediate:
961 OS << '<' << getImm();
962 if (getImmTy() != ImmTyNone) {
963 OS << " type: ";
964 printImmTy(OS, getImmTy());
965 }
966 OS << " mods: " << Imm.Mods << '>';
967 break;
968 case Token:
969 OS << '\'' << getToken() << '\'';
970 break;
971 case Expression:
972 OS << "<expr ";
973 MAI.printExpr(OS, *Expr);
974 OS << '>';
975 break;
976 }
977 }
978
979 static AMDGPUOperand::Ptr CreateImm(const AMDGPUAsmParser *AsmParser,
980 int64_t Val, SMLoc Loc,
981 ImmTy Type = ImmTyNone,
982 bool IsFPImm = false) {
983 auto Op = std::make_unique<AMDGPUOperand>(Immediate, AsmParser);
984 Op->Imm.Val = Val;
985 Op->Imm.IsFPImm = IsFPImm;
986 Op->Imm.Type = Type;
987 Op->Imm.Mods = Modifiers();
988 Op->StartLoc = Loc;
989 Op->EndLoc = Loc;
990 return Op;
991 }
992
993 static AMDGPUOperand::Ptr CreateToken(const AMDGPUAsmParser *AsmParser,
994 StringRef Str, SMLoc Loc,
995 bool HasExplicitEncodingSize = true) {
996 auto Res = std::make_unique<AMDGPUOperand>(Token, AsmParser);
997 Res->Tok.Data = Str.data();
998 Res->Tok.Length = Str.size();
999 Res->StartLoc = Loc;
1000 Res->EndLoc = Loc;
1001 return Res;
1002 }
1003
1004 static AMDGPUOperand::Ptr CreateReg(const AMDGPUAsmParser *AsmParser,
1005 MCRegister Reg, SMLoc S, SMLoc E) {
1006 auto Op = std::make_unique<AMDGPUOperand>(Register, AsmParser);
1007 Op->Reg.RegNo = Reg;
1008 Op->Reg.Mods = Modifiers();
1009 Op->StartLoc = S;
1010 Op->EndLoc = E;
1011 return Op;
1012 }
1013
1014 static AMDGPUOperand::Ptr CreateExpr(const AMDGPUAsmParser *AsmParser,
1015 const class MCExpr *Expr, SMLoc S) {
1016 auto Op = std::make_unique<AMDGPUOperand>(Expression, AsmParser);
1017 Op->Expr = Expr;
1018 Op->StartLoc = S;
1019 Op->EndLoc = S;
1020 return Op;
1021 }
1022};
1023
1024raw_ostream &operator<<(raw_ostream &OS, AMDGPUOperand::Modifiers Mods) {
1025 OS << "abs:" << Mods.Abs << " neg: " << Mods.Neg << " sext:" << Mods.Sext;
1026 return OS;
1027}
1028
1029//===----------------------------------------------------------------------===//
1030// AsmParser
1031//===----------------------------------------------------------------------===//
1032
1033// TODO: define GET_SUBTARGET_FEATURE_NAME
1034#define GET_REGISTER_MATCHER
1035#include "AMDGPUGenAsmMatcher.inc"
1036#undef GET_REGISTER_MATCHER
1037#undef GET_SUBTARGET_FEATURE_NAME
1038
1039// Holds info related to the current kernel, e.g. count of SGPRs used.
1040// Kernel scope begins at .amdgpu_hsa_kernel directive, ends at next
1041// .amdgpu_hsa_kernel or at EOF.
1042class KernelScopeInfo {
1043 int SgprIndexUnusedMin = -1;
1044 int VgprIndexUnusedMin = -1;
1045 int AgprIndexUnusedMin = -1;
1046 MCContext *Ctx = nullptr;
1047 MCSubtargetInfo const *MSTI = nullptr;
1048
1049 void usesSgprAt(int i) {
1050 if (i >= SgprIndexUnusedMin) {
1051 SgprIndexUnusedMin = ++i;
1052 if (Ctx) {
1053 MCSymbol *const Sym =
1054 Ctx->getOrCreateSymbol(Twine(".kernel.sgpr_count"));
1055 Sym->setVariableValue(MCConstantExpr::create(SgprIndexUnusedMin, *Ctx));
1056 }
1057 }
1058 }
1059
1060 void usesVgprAt(int i) {
1061 if (i >= VgprIndexUnusedMin) {
1062 VgprIndexUnusedMin = ++i;
1063 if (Ctx) {
1064 MCSymbol *const Sym =
1065 Ctx->getOrCreateSymbol(Twine(".kernel.vgpr_count"));
1066 int totalVGPR = getTotalNumVGPRs(isGFX90A(*MSTI), AgprIndexUnusedMin,
1067 VgprIndexUnusedMin);
1068 Sym->setVariableValue(MCConstantExpr::create(totalVGPR, *Ctx));
1069 }
1070 }
1071 }
1072
1073 void usesAgprAt(int i) {
1074 // Instruction will error in AMDGPUAsmParser::matchAndEmitInstruction
1075 if (!hasMAIInsts(*MSTI))
1076 return;
1077
1078 if (i >= AgprIndexUnusedMin) {
1079 AgprIndexUnusedMin = ++i;
1080 if (Ctx) {
1081 MCSymbol *const Sym =
1082 Ctx->getOrCreateSymbol(Twine(".kernel.agpr_count"));
1083 Sym->setVariableValue(MCConstantExpr::create(AgprIndexUnusedMin, *Ctx));
1084
1085 // Also update vgpr_count (dependent on agpr_count for gfx908/gfx90a)
1086 MCSymbol *const vSym =
1087 Ctx->getOrCreateSymbol(Twine(".kernel.vgpr_count"));
1088 int totalVGPR = getTotalNumVGPRs(isGFX90A(*MSTI), AgprIndexUnusedMin,
1089 VgprIndexUnusedMin);
1090 vSym->setVariableValue(MCConstantExpr::create(totalVGPR, *Ctx));
1091 }
1092 }
1093 }
1094
1095public:
1096 KernelScopeInfo() = default;
1097
1098 void initialize(MCContext &Context) {
1099 Ctx = &Context;
1100 MSTI = Ctx->getSubtargetInfo();
1101
1102 usesSgprAt(SgprIndexUnusedMin = -1);
1103 usesVgprAt(VgprIndexUnusedMin = -1);
1104 if (hasMAIInsts(*MSTI)) {
1105 usesAgprAt(AgprIndexUnusedMin = -1);
1106 }
1107 }
1108
1109 void usesRegister(RegisterKind RegKind, unsigned DwordRegIndex,
1110 unsigned RegWidth) {
1111 switch (RegKind) {
1112 case IS_SGPR:
1113 usesSgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1114 break;
1115 case IS_AGPR:
1116 usesAgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1117 break;
1118 case IS_VGPR:
1119 usesVgprAt(DwordRegIndex + divideCeil(RegWidth, 32) - 1);
1120 break;
1121 default:
1122 break;
1123 }
1124 }
1125};
1126
1127class AMDGPUAsmParser : public MCTargetAsmParser {
1128 MCAsmParser &Parser;
1129
1130 unsigned ForcedEncodingSize = 0;
1131 bool ForcedDPP = false;
1132 bool ForcedSDWA = false;
1133 KernelScopeInfo KernelScope;
1134 const unsigned HwMode;
1135 const AMDGPU::GPUKind Gfx;
1136 const AMDGPU::IsaVersion ISA;
1137
1138 /// @name Auto-generated Match Functions
1139 /// {
1140
1141#define GET_ASSEMBLER_HEADER
1142#include "AMDGPUGenAsmMatcher.inc"
1143
1144 /// }
1145
1146 /// Get size of register operand
1147 unsigned getRegOperandSize(const MCInstrDesc &Desc, unsigned OpNo) const {
1148 assert(OpNo < Desc.NumOperands);
1149 int16_t RCID = MII.getOpRegClassID(Desc.operands()[OpNo], HwMode);
1150 return getRegBitWidth(RCID) / 8;
1151 }
1152
1153 std::optional<AMDGPU::InfoSectionData> InfoData;
1154
1155 /// Whether the leading .amdgcn_target directive has been emitted to the
1156 /// output streamer yet. The emission is deferred until the first piece of
1157 /// content (instruction or kernel descriptor) so that any leading
1158 /// .amdgcn_target/.amd_amdgpu_isa directive in the source has had a chance to
1159 /// update the target ID first.
1160 bool TargetDirectiveEmitted = false;
1161
1162 /// State for checking that every kernel named in a .amdhsa_kernel directive
1163 /// begins with the required prologue instruction sequence. Because the
1164 /// directive may appear either before or after the kernel's label (it is
1165 /// normally emitted after the function body, in .rodata), validation is
1166 /// deferred to onEndOfFile(). We record an order-independent timeline of
1167 /// parsed labels and emitted instruction opcodes, plus the set of symbols
1168 /// named by .amdhsa_kernel directives, and match them up at end of file.
1169 SmallVector<unsigned> OpcodeStream;
1171 OpcodeStreamSymbols;
1172 SmallPtrSet<const MCSymbol *, 8> AMDHSAKernelSymbols;
1173
1174 /// Verify recorded kernel prologues.
1175 void checkKernelPrologues();
1176
1177private:
1178 void createConstantSymbol(StringRef Id, int64_t Val);
1179
1180 bool ParseAsAbsoluteExpression(uint32_t &Ret);
1181 bool OutOfRangeError(SMRange Range);
1182 /// Calculate VGPR/SGPR blocks required for given target, reserved
1183 /// registers, and user-specified NextFreeXGPR values.
1184 ///
1185 /// \param Features [in] Target features, used for bug corrections.
1186 /// \param VCCUsed [in] Whether VCC special SGPR is reserved.
1187 /// \param FlatScrUsed [in] Whether FLAT_SCRATCH special SGPR is reserved.
1188 /// \param XNACKUsed [in] Whether XNACK_MASK special SGPR is reserved.
1189 /// \param EnableWavefrontSize32 [in] Value of ENABLE_WAVEFRONT_SIZE32 kernel
1190 /// descriptor field, if valid.
1191 /// \param NextFreeVGPR [in] Max VGPR number referenced, plus one.
1192 /// \param VGPRRange [in] Token range, used for VGPR diagnostics.
1193 /// \param NextFreeSGPR [in] Max SGPR number referenced, plus one.
1194 /// \param SGPRRange [in] Token range, used for SGPR diagnostics.
1195 /// \param VGPRBlocks [out] Result VGPR block count.
1196 /// \param SGPRBlocks [out] Result SGPR block count.
1197 bool calculateGPRBlocks(const FeatureBitset &Features, const MCExpr *VCCUsed,
1198 const MCExpr *FlatScrUsed, bool XNACKUsed,
1199 std::optional<bool> EnableWavefrontSize32,
1200 const MCExpr *NextFreeVGPR, SMRange VGPRRange,
1201 const MCExpr *NextFreeSGPR, SMRange SGPRRange,
1202 const MCExpr *&VGPRBlocks, const MCExpr *&SGPRBlocks);
1203 bool ParseDirectiveAMDGCNTarget();
1204 bool ParseDirectiveAMDHSACodeObjectVersion();
1205 bool ParseDirectiveAMDHSAKernel();
1206 bool ParseAMDKernelCodeTValue(StringRef ID, AMDGPUMCKernelCodeT &Header);
1207 bool ParseDirectiveAMDKernelCodeT();
1208 // TODO: Possibly make subtargetHasRegister const.
1209 bool subtargetHasRegister(const MCRegisterInfo &MRI, MCRegister Reg);
1210 bool ParseDirectiveAMDGPUHsaKernel();
1211
1212 bool ParseDirectiveISAVersion();
1213 bool ParseDirectiveHSAMetadata();
1214 bool ParseDirectivePALMetadataBegin();
1215 bool ParseDirectivePALMetadata();
1216 bool ParseDirectiveAMDGPULDS();
1217 bool ParseDirectiveAMDGPUInfo();
1218
1219 /// Common code to parse out a block of text (typically YAML) between start
1220 /// and end directives.
1221 bool ParseToEndDirective(const char *AssemblerDirectiveBegin,
1222 const char *AssemblerDirectiveEnd,
1223 std::string &CollectString);
1224
1225 bool AddNextRegisterToList(MCRegister &Reg, unsigned &RegWidth,
1226 RegisterKind RegKind, MCRegister Reg1,
1227 RegisterKind RegKind1, SMLoc Loc);
1228 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &Reg,
1229 unsigned &RegNum, unsigned &RegWidth,
1230 bool RestoreOnFailure = false);
1231 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &Reg,
1232 unsigned &RegNum, unsigned &RegWidth,
1233 SmallVectorImpl<AsmToken> &Tokens);
1234 MCRegister ParseRegularReg(RegisterKind &RegKind, unsigned &RegNum,
1235 unsigned &RegWidth,
1236 SmallVectorImpl<AsmToken> &Tokens);
1237 MCRegister ParseSpecialReg(RegisterKind &RegKind, unsigned &RegNum,
1238 unsigned &RegWidth,
1239 SmallVectorImpl<AsmToken> &Tokens);
1240 MCRegister ParseRegList(RegisterKind &RegKind, unsigned &RegNum,
1241 unsigned &RegWidth,
1242 SmallVectorImpl<AsmToken> &Tokens);
1243 bool ParseRegRange(unsigned &Num, unsigned &Width, unsigned &SubReg);
1244 MCRegister getRegularReg(RegisterKind RegKind, unsigned RegNum,
1245 unsigned SubReg, unsigned RegWidth, SMLoc Loc);
1246
1247 bool isRegister();
1248 bool isRegister(const AsmToken &Token, const AsmToken &NextToken) const;
1249 std::optional<StringRef> getGprCountSymbolName(RegisterKind RegKind);
1250 void initializeGprCountSymbol(RegisterKind RegKind);
1251 bool updateGprCountSymbols(RegisterKind RegKind, unsigned DwordRegIndex,
1252 unsigned RegWidth);
1253 void cvtMubufImpl(MCInst &Inst, const OperandVector &Operands, bool IsAtomic);
1254
1255public:
1256 enum OperandMode {
1257 OperandMode_Default,
1258 OperandMode_NSA,
1259 };
1260
1261 using OptionalImmIndexMap = std::map<AMDGPUOperand::ImmTy, unsigned>;
1262
1263 AMDGPUAsmParser(const MCSubtargetInfo &STI, MCAsmParser &_Parser,
1264 const MCInstrInfo &MII)
1265 : MCTargetAsmParser(STI, MII), Parser(_Parser),
1266 HwMode(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
1267 Gfx(AMDGPU::parseArchAMDGCN(STI.getCPU())),
1268 ISA(AMDGPU::getIsaVersion(STI.getCPU())) {
1270
1271 setAvailableFeatures(ComputeAvailableFeatures(getFeatureBits()));
1272
1273 if (ISA.Major >= 6 && isHsaAbi(getSTI())) {
1274 createConstantSymbol(".amdgcn.gfx_generation_number", ISA.Major);
1275 createConstantSymbol(".amdgcn.gfx_generation_minor", ISA.Minor);
1276 createConstantSymbol(".amdgcn.gfx_generation_stepping", ISA.Stepping);
1277 } else {
1278 createConstantSymbol(".option.machine_version_major", ISA.Major);
1279 createConstantSymbol(".option.machine_version_minor", ISA.Minor);
1280 createConstantSymbol(".option.machine_version_stepping", ISA.Stepping);
1281 }
1282 if (ISA.Major >= 6 && isHsaAbi(getSTI())) {
1283 initializeGprCountSymbol(IS_VGPR);
1284 initializeGprCountSymbol(IS_SGPR);
1285 } else
1286 KernelScope.initialize(getContext());
1287
1288 for (auto [Symbol, Code] : AMDGPU::UCVersion::getGFXVersions())
1289 createConstantSymbol(Symbol, Code);
1290
1291 createConstantSymbol("UC_VERSION_W64_BIT", 0x2000);
1292 createConstantSymbol("UC_VERSION_W32_BIT", 0x4000);
1293 createConstantSymbol("UC_VERSION_MDP_BIT", 0x8000);
1294 }
1295
1296 bool hasMIMG_R128() const { return AMDGPU::hasMIMG_R128(getSTI()); }
1297
1298 bool hasPackedD16() const { return AMDGPU::hasPackedD16(getSTI()); }
1299
1300 bool hasA16() const { return AMDGPU::hasA16(getSTI()); }
1301
1302 bool hasG16() const { return AMDGPU::hasG16(getSTI()); }
1303
1304 bool hasGDS() const { return AMDGPU::hasGDS(getSTI()); }
1305
1306 bool isSI() const { return AMDGPU::isSI(getSTI()); }
1307
1308 bool isCI() const { return AMDGPU::isCI(getSTI()); }
1309
1310 bool isVI() const { return AMDGPU::isVI(getSTI()); }
1311
1312 bool isGFX9() const { return AMDGPU::isGFX9(getSTI()); }
1313
1314 // TODO: isGFX90A is also true for GFX940. We need to clean it.
1315 bool isGFX90A() const { return AMDGPU::isGFX90A(getSTI()); }
1316
1317 bool isGFX940() const { return AMDGPU::isGFX940(getSTI()); }
1318
1319 bool isGFX9Plus() const { return AMDGPU::isGFX9Plus(getSTI()); }
1320
1321 bool isGFX10Plus() const { return AMDGPU::isGFX10Plus(getSTI()); }
1322
1323 bool isGFX11() const { return AMDGPU::isGFX11(getSTI()); }
1324
1325 bool isGFX11Plus() const { return AMDGPU::isGFX11Plus(getSTI()); }
1326
1327 bool isGFX12() const { return AMDGPU::isGFX12(getSTI()); }
1328
1329 bool isGFX12Plus() const { return AMDGPU::isGFX12Plus(getSTI()); }
1330
1331 bool isGFX1250Plus() const { return AMDGPU::isGFX1250Plus(getSTI()); }
1332
1333 bool hasBVHRayTracingInsts() const {
1334 return getFeatureBits()[AMDGPU::FeatureBVHRayTracingInsts];
1335 }
1336
1337 bool isWave32() const { return getAvailableFeatures()[Feature_isWave32Bit]; }
1338
1339 bool isWave64() const { return getAvailableFeatures()[Feature_isWave64Bit]; }
1340
1341 bool hasInv2PiInlineImm() const {
1342 return getFeatureBits()[AMDGPU::FeatureInv2PiInlineImm];
1343 }
1344
1345 bool has64BitLiterals() const {
1346 return getFeatureBits()[AMDGPU::Feature64BitLiterals];
1347 }
1348
1349 bool hasFlatOffsets() const {
1350 return getFeatureBits()[AMDGPU::FeatureFlatInstOffsets];
1351 }
1352
1353 bool hasTrue16Insts() const {
1354 return getFeatureBits()[AMDGPU::FeatureTrue16BitInsts];
1355 }
1356
1357 bool hasArchitectedFlatScratch() const {
1358 return getFeatureBits()[AMDGPU::FeatureArchitectedFlatScratch];
1359 }
1360
1361 bool hasSGPR102_SGPR103() const { return !isVI() && !isGFX9(); }
1362
1363 bool hasSGPR104_SGPR105() const { return isGFX10Plus(); }
1364
1365 bool hasIntClamp() const { return getFeatureBits()[AMDGPU::FeatureIntClamp]; }
1366
1367 bool hasPartialNSAEncoding() const {
1368 return getFeatureBits()[AMDGPU::FeaturePartialNSAEncoding];
1369 }
1370
1371 bool hasGloballyAddressableScratch() const {
1372 return getFeatureBits()[AMDGPU::FeatureGloballyAddressableScratch];
1373 }
1374
1375 unsigned getNSAMaxSize(bool HasSampler = false) const {
1376 return AMDGPU::getNSAMaxSize(getSTI(), HasSampler);
1377 }
1378
1379 unsigned getMaxNumUserSGPRs() const {
1380 return AMDGPU::getMaxNumUserSGPRs(getSTI());
1381 }
1382
1383 bool hasKernargPreload() const { return AMDGPU::hasKernargPreload(getSTI()); }
1384
1385 AMDGPUTargetStreamer &getTargetStreamer() {
1386 MCTargetStreamer &TS = *getParser().getStreamer().getTargetStreamer();
1387 return static_cast<AMDGPUTargetStreamer &>(TS);
1388 }
1389
1390 MCContext &getContext() const {
1391 // We need this const_cast because for some reason getContext() is not const
1392 // in MCAsmParser.
1393 return const_cast<AMDGPUAsmParser *>(this)->MCTargetAsmParser::getContext();
1394 }
1395
1396 const MCRegisterInfo *getMRI() const {
1397 return getContext().getRegisterInfo();
1398 }
1399
1400 const MCInstrInfo *getMII() const { return &MII; }
1401
1402 // Resolve a RegClassByHwModeUses index to a register class id for the active
1403 // HwMode; -1 if the mode has no entry.
1404 int16_t getTargetRegClass(unsigned TargetRCIdx) const {
1405 return MII.getRegClassByHwModeTable(HwMode)[TargetRCIdx];
1406 }
1407
1408 // FIXME: This should not be used. Instead, should use queries derived from
1409 // getAvailableFeatures().
1410 const FeatureBitset &getFeatureBits() const {
1411 return getSTI().getFeatureBits();
1412 }
1413
1414 void setForcedEncodingSize(unsigned Size) { ForcedEncodingSize = Size; }
1415 void setForcedDPP(bool ForceDPP_) { ForcedDPP = ForceDPP_; }
1416 void setForcedSDWA(bool ForceSDWA_) { ForcedSDWA = ForceSDWA_; }
1417
1418 unsigned getForcedEncodingSize() const { return ForcedEncodingSize; }
1419 bool isForcedVOP3() const { return ForcedEncodingSize == 64; }
1420 bool isForcedDPP() const { return ForcedDPP; }
1421 bool isForcedSDWA() const { return ForcedSDWA; }
1422 ArrayRef<unsigned> getMatchedVariants() const;
1423 StringRef getMatchedVariantName() const;
1424
1425 std::unique_ptr<AMDGPUOperand> parseRegister(bool RestoreOnFailure = false);
1426 bool ParseRegister(MCRegister &RegNo, SMLoc &StartLoc, SMLoc &EndLoc,
1427 bool RestoreOnFailure);
1428 bool parseRegister(MCRegister &Reg, SMLoc &StartLoc, SMLoc &EndLoc) override;
1429 ParseStatus tryParseRegister(MCRegister &Reg, SMLoc &StartLoc,
1430 SMLoc &EndLoc) override;
1431 unsigned checkTargetMatchPredicate(MCInst &Inst) override;
1432 unsigned validateTargetOperandClass(MCParsedAsmOperand &Op,
1433 unsigned Kind) override;
1434 bool matchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode,
1435 OperandVector &Operands, MCStreamer &Out,
1436 uint64_t &ErrorInfo,
1437 bool MatchingInlineAsm) override;
1438 bool ParseDirective(AsmToken DirectiveID) override;
1439 void doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) override;
1440 void onEndOfFile() override;
1441 ParseStatus parseOperand(OperandVector &Operands, StringRef Mnemonic,
1442 OperandMode Mode = OperandMode_Default);
1443 StringRef parseMnemonicSuffix(StringRef Name);
1444 bool parseInstruction(ParseInstructionInfo &Info, StringRef Name,
1445 SMLoc NameLoc, OperandVector &Operands) override;
1446 // bool ProcessInstruction(MCInst &Inst);
1447
1448 ParseStatus parseTokenOp(StringRef Name, OperandVector &Operands);
1449
1450 ParseStatus parseIntWithPrefix(const char *Prefix, int64_t &Int);
1451
1452 ParseStatus
1453 parseIntWithPrefix(const char *Prefix, OperandVector &Operands,
1454 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1455 std::function<bool(int64_t &)> ConvertResult = nullptr);
1456
1457 ParseStatus parseOperandArrayWithPrefix(
1458 const char *Prefix, OperandVector &Operands,
1459 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1460 bool (*ConvertResult)(int64_t &) = nullptr);
1461
1462 ParseStatus
1463 parseNamedBit(StringRef Name, OperandVector &Operands,
1464 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1465 bool IgnoreNegative = false);
1466 unsigned getCPolKind(StringRef Id, StringRef Mnemo, bool &Disabling) const;
1467 ParseStatus parseCPol(OperandVector &Operands);
1468 ParseStatus parseScope(OperandVector &Operands, int64_t &Scope);
1469 ParseStatus parseTH(OperandVector &Operands, int64_t &TH);
1470 ParseStatus parseStringWithPrefix(StringRef Prefix, StringRef &Value,
1471 SMLoc &StringLoc);
1472 ParseStatus parseStringOrIntWithPrefix(OperandVector &Operands,
1473 StringRef Name,
1474 ArrayRef<const char *> Ids,
1475 int64_t &IntVal);
1476 ParseStatus parseStringOrIntWithPrefix(OperandVector &Operands,
1477 StringRef Name,
1478 ArrayRef<const char *> Ids,
1479 AMDGPUOperand::ImmTy Type);
1480
1481 bool isModifier();
1482 bool isOperandModifier(const AsmToken &Token,
1483 const AsmToken &NextToken) const;
1484 bool isRegOrOperandModifier(const AsmToken &Token,
1485 const AsmToken &NextToken) const;
1486 bool isNamedOperandModifier(const AsmToken &Token,
1487 const AsmToken &NextToken) const;
1488 bool isOpcodeModifierWithVal(const AsmToken &Token,
1489 const AsmToken &NextToken) const;
1490 bool parseSP3NegModifier();
1491 ParseStatus parseImm(OperandVector &Operands, bool HasSP3AbsModifier = false,
1492 LitModifier Lit = LitModifier::None);
1493 ParseStatus parseReg(OperandVector &Operands);
1494 ParseStatus parseRegOrImm(OperandVector &Operands, bool HasSP3AbsMod = false,
1495 LitModifier Lit = LitModifier::None);
1496 ParseStatus parseRegOrImmWithFPInputMods(OperandVector &Operands,
1497 bool AllowImm = true);
1498 ParseStatus parseRegOrImmWithIntInputMods(OperandVector &Operands,
1499 bool AllowImm = true);
1500 ParseStatus parseRegWithFPInputMods(OperandVector &Operands);
1501 ParseStatus parseRegWithIntInputMods(OperandVector &Operands);
1502 ParseStatus parseRsrcReg(OperandVector &Operands);
1503 ParseStatus parseVReg32OrOff(OperandVector &Operands);
1504 ParseStatus tryParseIndexKey(OperandVector &Operands,
1505 AMDGPUOperand::ImmTy ImmTy);
1506 ParseStatus parseIndexKey8bit(OperandVector &Operands);
1507 ParseStatus parseIndexKey16bit(OperandVector &Operands);
1508 ParseStatus parseIndexKey32bit(OperandVector &Operands);
1509 ParseStatus tryParseMatrixFMT(OperandVector &Operands, StringRef Name,
1510 AMDGPUOperand::ImmTy Type);
1511 ParseStatus parseMatrixAFMT(OperandVector &Operands);
1512 ParseStatus parseMatrixBFMT(OperandVector &Operands);
1513 ParseStatus tryParseMatrixScale(OperandVector &Operands, StringRef Name,
1514 AMDGPUOperand::ImmTy Type);
1515 ParseStatus parseMatrixAScale(OperandVector &Operands);
1516 ParseStatus parseMatrixBScale(OperandVector &Operands);
1517 ParseStatus tryParseMatrixScaleFmt(OperandVector &Operands, StringRef Name,
1518 AMDGPUOperand::ImmTy Type);
1519 ParseStatus parseMatrixAScaleFmt(OperandVector &Operands);
1520 ParseStatus parseMatrixBScaleFmt(OperandVector &Operands);
1521
1522 ParseStatus parseDfmtNfmt(int64_t &Format);
1523 ParseStatus parseUfmt(int64_t &Format);
1524 ParseStatus parseSymbolicSplitFormat(StringRef FormatStr, SMLoc Loc,
1525 int64_t &Format);
1526 ParseStatus parseSymbolicUnifiedFormat(StringRef FormatStr, SMLoc Loc,
1527 int64_t &Format);
1528 ParseStatus parseFORMAT(OperandVector &Operands);
1529 ParseStatus parseSymbolicOrNumericFormat(int64_t &Format);
1530 ParseStatus parseNumericFormat(int64_t &Format);
1531 ParseStatus parseFlatOffset(OperandVector &Operands);
1532 ParseStatus parseR128A16(OperandVector &Operands);
1533 ParseStatus parseBLGP(OperandVector &Operands);
1534 bool tryParseFmt(const char *Pref, int64_t MaxVal, int64_t &Val);
1535 bool matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt, StringRef FormatStr,
1536 SMLoc Loc);
1537
1538 void cvtExp(MCInst &Inst, const OperandVector &Operands);
1539
1540 bool parseCnt(int64_t &IntVal);
1541 ParseStatus parseSWaitCnt(OperandVector &Operands);
1542
1543 bool parseDepCtr(int64_t &IntVal, unsigned &Mask);
1544 void depCtrError(SMLoc Loc, int ErrorId, StringRef DepCtrName);
1545 ParseStatus parseDepCtr(OperandVector &Operands);
1546
1547 bool parseDelay(int64_t &Delay);
1548 ParseStatus parseSDelayALU(OperandVector &Operands);
1549
1550 ParseStatus parseHwreg(OperandVector &Operands);
1551
1552private:
1553 struct OperandInfoTy {
1554 SMLoc Loc;
1555 int64_t Val;
1556 bool IsSymbolic = false;
1557 bool IsDefined = false;
1558
1559 constexpr OperandInfoTy(int64_t Val) : Val(Val) {}
1560 };
1561
1562 struct StructuredOpField : OperandInfoTy {
1563 StringLiteral Id;
1564 StringLiteral Desc;
1565 unsigned Width;
1566 bool IsDefined = false;
1567
1568 constexpr StructuredOpField(StringLiteral Id, StringLiteral Desc,
1569 unsigned Width, int64_t Default)
1570 : OperandInfoTy(Default), Id(Id), Desc(Desc), Width(Width) {}
1571 virtual ~StructuredOpField() = default;
1572
1573 bool Error(AMDGPUAsmParser &Parser, const Twine &Err) const {
1574 Parser.Error(Loc, "invalid " + Desc + ": " + Err);
1575 return false;
1576 }
1577
1578 virtual bool validate(AMDGPUAsmParser &Parser) const {
1579 if (IsSymbolic && Val == OPR_ID_UNSUPPORTED)
1580 return Error(Parser, "not supported on this GPU");
1581 if (!isUIntN(Width, Val))
1582 return Error(Parser, "only " + Twine(Width) + "-bit values are legal");
1583 return true;
1584 }
1585 };
1586
1587 ParseStatus parseStructuredOpFields(ArrayRef<StructuredOpField *> Fields);
1588 bool validateStructuredOpFields(ArrayRef<const StructuredOpField *> Fields);
1589
1590 bool parseSendMsgBody(OperandInfoTy &Msg, OperandInfoTy &Op,
1591 OperandInfoTy &Stream);
1592 bool validateSendMsg(const OperandInfoTy &Msg, const OperandInfoTy &Op,
1593 const OperandInfoTy &Stream);
1594
1595 ParseStatus parseHwregFunc(OperandInfoTy &HwReg, OperandInfoTy &Offset,
1596 OperandInfoTy &Width);
1597
1598 const AMDGPUOperand &findMCOperand(const OperandVector &Operands,
1599 int MCOpIdx) const;
1600
1601 static SMLoc getLaterLoc(SMLoc a, SMLoc b);
1602
1603 SMLoc getFlatOffsetLoc(const OperandVector &Operands) const;
1604 SMLoc getSMEMOffsetLoc(const OperandVector &Operands) const;
1605 SMLoc getBLGPLoc(const OperandVector &Operands) const;
1606
1607 SMLoc getOperandLoc(const OperandVector &Operands, int MCOpIdx) const;
1608 SMLoc getOperandLoc(std::function<bool(const AMDGPUOperand &)> Test,
1609 const OperandVector &Operands) const;
1610 SMLoc getImmLoc(AMDGPUOperand::ImmTy Type,
1611 const OperandVector &Operands) const;
1612 SMLoc getInstLoc(const OperandVector &Operands) const;
1613
1614 bool validateInstruction(const MCInst &Inst, SMLoc IDLoc,
1615 const OperandVector &Operands);
1616 bool validateOffset(const MCInst &Inst, const OperandVector &Operands);
1617 bool validateFlatOffset(const MCInst &Inst, const OperandVector &Operands);
1618 bool validateSMEMOffset(const MCInst &Inst, const OperandVector &Operands);
1619 bool validateBF16InlineConst(const MCInst &Inst,
1620 const OperandVector &Operands);
1621 bool validateSOPLiteral(const MCInst &Inst, const OperandVector &Operands);
1622 bool validateConstantBusLimitations(const MCInst &Inst,
1623 const OperandVector &Operands);
1624 std::optional<unsigned> checkVOPDRegBankConstraints(const MCInst &Inst,
1625 bool AsVOPD3);
1626 bool validateVOPD(const MCInst &Inst, const OperandVector &Operands);
1627 bool tryVOPD(const MCInst &Inst);
1628 bool tryVOPD3(const MCInst &Inst);
1629 bool tryAnotherVOPDEncoding(const MCInst &Inst);
1630
1631 bool validateIntClampSupported(const MCInst &Inst);
1632 bool validateMIMGAtomicDMask(const MCInst &Inst);
1633 bool validateMIMGGatherDMask(const MCInst &Inst);
1634 bool validateMovrels(const MCInst &Inst, const OperandVector &Operands);
1635 bool validateMIMGDataSize(const MCInst &Inst, SMLoc IDLoc);
1636 bool validateMIMGAddrSize(const MCInst &Inst, SMLoc IDLoc);
1637 bool validateMIMGD16(const MCInst &Inst);
1638 bool validateMIMGDim(const MCInst &Inst, const OperandVector &Operands);
1639 bool validateTensorR128(const MCInst &Inst);
1640 bool validateMIMGMSAA(const MCInst &Inst);
1641 bool validateOpSel(const MCInst &Inst);
1642 bool validateTrue16OpSel(const MCInst &Inst);
1643 bool validateNeg(const MCInst &Inst, AMDGPU::OpName OpName);
1644 bool validateDPP(const MCInst &Inst, const OperandVector &Operands);
1645 bool validateVccOperand(MCRegister Reg) const;
1646 bool validateVOPLiteral(const MCInst &Inst, const OperandVector &Operands);
1647 bool validateMAIAccWrite(const MCInst &Inst, const OperandVector &Operands);
1648 bool validateMAISrc2(const MCInst &Inst, const OperandVector &Operands);
1649 bool validateMFMA(const MCInst &Inst, const OperandVector &Operands);
1650 bool validateAGPRLdSt(const MCInst &Inst) const;
1651 bool validateVGPRAlign(const MCInst &Inst) const;
1652 bool validateBLGP(const MCInst &Inst, const OperandVector &Operands);
1653 bool validateDS(const MCInst &Inst, const OperandVector &Operands);
1654 bool validateGWS(const MCInst &Inst, const OperandVector &Operands);
1655 bool validateDivScale(const MCInst &Inst);
1656 bool validateWaitCnt(const MCInst &Inst, const OperandVector &Operands);
1657 bool validateCoherencyBits(const MCInst &Inst, const OperandVector &Operands,
1658 SMLoc IDLoc);
1659 bool validateTHAndScopeBits(const MCInst &Inst, const OperandVector &Operands,
1660 const unsigned CPol);
1661 bool validateTFE(const MCInst &Inst, const OperandVector &Operands);
1662 bool validateLdsDirect(const MCInst &Inst, const OperandVector &Operands);
1663 bool validateWMMA(const MCInst &Inst, const OperandVector &Operands);
1664 bool validateMonitorSleep(const MCInst &Inst, const OperandVector &Operands);
1665 bool validateClusterBarrierIsFirst(const MCInst &Inst,
1666 const OperandVector &Operands);
1667 bool validateScaleSel(const MCInst &Inst, const OperandVector &Operands);
1668 unsigned getConstantBusLimit(unsigned Opcode) const;
1669 bool usesConstantBus(const MCInst &Inst, unsigned OpIdx);
1670 bool isInlineConstant(const MCInst &Inst, unsigned OpIdx) const;
1671 MCRegister findImplicitSGPRReadInVOP(const MCInst &Inst) const;
1672
1673 bool isSupportedMnemo(StringRef Mnemo, const FeatureBitset &FBS);
1674 bool isSupportedMnemo(StringRef Mnemo, const FeatureBitset &FBS,
1675 ArrayRef<unsigned> Variants);
1676 bool checkUnsupportedInstruction(StringRef Name, SMLoc IDLoc);
1677
1678 bool isId(const StringRef Id) const;
1679 bool isId(const AsmToken &Token, const StringRef Id) const;
1680 bool isToken(const AsmToken::TokenKind Kind) const;
1681 StringRef getId() const;
1682 bool trySkipId(const StringRef Id);
1683 bool trySkipId(const StringRef Pref, const StringRef Id);
1684 bool trySkipId(const StringRef Id, const AsmToken::TokenKind Kind);
1685 bool trySkipToken(const AsmToken::TokenKind Kind);
1686 bool skipToken(const AsmToken::TokenKind Kind, const StringRef ErrMsg);
1687 bool parseString(StringRef &Val,
1688 const StringRef ErrMsg = "expected a string");
1689 bool parseId(StringRef &Val, const StringRef ErrMsg = "");
1690
1691 void peekTokens(MutableArrayRef<AsmToken> Tokens);
1692 AsmToken::TokenKind getTokenKind() const;
1693 bool parseExpr(int64_t &Imm, StringRef Expected = "");
1695 StringRef getTokenStr() const;
1696 AsmToken peekToken(bool ShouldSkipSpace = true);
1697 AsmToken getToken() const;
1698 SMLoc getLoc() const;
1699 void lex();
1700
1701public:
1702 void onBeginOfFile() override;
1703 /// Emit the deferred leading .amdgcn_target directive if it has not been
1704 /// emitted yet. Called before emitting the first instruction or kernel
1705 /// descriptor.
1706 void emitTargetDirective();
1707 bool parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) override;
1708
1709 ParseStatus parseCustomOperand(OperandVector &Operands, unsigned MCK);
1710
1711 ParseStatus parseExpTgt(OperandVector &Operands);
1712 ParseStatus parseSendMsg(OperandVector &Operands);
1713 ParseStatus parseWaitEvent(OperandVector &Operands);
1714 ParseStatus parseInterpSlot(OperandVector &Operands);
1715 ParseStatus parseInterpAttr(OperandVector &Operands);
1716 ParseStatus parseSOPPBrTarget(OperandVector &Operands);
1717 ParseStatus parseBoolReg(OperandVector &Operands);
1718
1719 bool parseSwizzleOperand(int64_t &Op, const unsigned MinVal,
1720 const unsigned MaxVal, const Twine &ErrMsg,
1721 SMLoc &Loc);
1722 bool parseSwizzleOperands(const unsigned OpNum, int64_t *Op,
1723 const unsigned MinVal, const unsigned MaxVal,
1724 const StringRef ErrMsg);
1725 ParseStatus parseSwizzle(OperandVector &Operands);
1726 bool parseSwizzleOffset(int64_t &Imm);
1727 bool parseSwizzleMacro(int64_t &Imm);
1728 bool parseSwizzleQuadPerm(int64_t &Imm);
1729 bool parseSwizzleBitmaskPerm(int64_t &Imm);
1730 bool parseSwizzleBroadcast(int64_t &Imm);
1731 bool parseSwizzleSwap(int64_t &Imm);
1732 bool parseSwizzleReverse(int64_t &Imm);
1733 bool parseSwizzleFFT(int64_t &Imm);
1734 bool parseSwizzleRotate(int64_t &Imm);
1735
1736 ParseStatus parseGPRIdxMode(OperandVector &Operands);
1737 int64_t parseGPRIdxMacro();
1738
1739 void cvtMubuf(MCInst &Inst, const OperandVector &Operands) {
1740 cvtMubufImpl(Inst, Operands, false);
1741 }
1742 void cvtMubufAtomic(MCInst &Inst, const OperandVector &Operands) {
1743 cvtMubufImpl(Inst, Operands, true);
1744 }
1745
1746 ParseStatus parseOModSI(OperandVector &Operands);
1747
1748 void cvtVOP3(MCInst &Inst, const OperandVector &Operands,
1749 OptionalImmIndexMap &OptionalIdx);
1750 void cvtScaledMFMA(MCInst &Inst, const OperandVector &Operands);
1751 void cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands);
1752 void cvtVOP3(MCInst &Inst, const OperandVector &Operands);
1753 void cvtVOP3P(MCInst &Inst, const OperandVector &Operands);
1754 void cvtSWMMAC(MCInst &Inst, const OperandVector &Operands);
1755
1756 void cvtVOPD(MCInst &Inst, const OperandVector &Operands);
1757 void cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands,
1758 OptionalImmIndexMap &OptionalIdx);
1759 void cvtVOP3P(MCInst &Inst, const OperandVector &Operands,
1760 OptionalImmIndexMap &OptionalIdx);
1761
1762 void cvtVOP3Interp(MCInst &Inst, const OperandVector &Operands);
1763 void cvtVINTERP(MCInst &Inst, const OperandVector &Operands);
1764 void cvtOpSelHelper(MCInst &Inst, unsigned OpSel);
1765
1766 bool parseDimId(unsigned &Encoding);
1767 ParseStatus parseDim(OperandVector &Operands);
1768 bool convertDppBoundCtrl(int64_t &BoundCtrl);
1769 ParseStatus parseDPP8(OperandVector &Operands);
1770 ParseStatus parseDPPCtrl(OperandVector &Operands);
1771 bool isSupportedDPPCtrl(StringRef Ctrl, const OperandVector &Operands);
1772 int64_t parseDPPCtrlSel(StringRef Ctrl);
1773 int64_t parseDPPCtrlPerm();
1774 void cvtDPP(MCInst &Inst, const OperandVector &Operands, bool IsDPP8 = false);
1775 void cvtDPP8(MCInst &Inst, const OperandVector &Operands) {
1776 cvtDPP(Inst, Operands, true);
1777 }
1778 void cvtVOP3DPP(MCInst &Inst, const OperandVector &Operands,
1779 bool IsDPP8 = false);
1780 void cvtVOP3DPP8(MCInst &Inst, const OperandVector &Operands) {
1781 cvtVOP3DPP(Inst, Operands, true);
1782 }
1783
1784 ParseStatus parseSDWASel(OperandVector &Operands, StringRef Prefix,
1785 AMDGPUOperand::ImmTy Type);
1786 ParseStatus parseSDWADstUnused(OperandVector &Operands);
1787 void cvtSdwaVOP1(MCInst &Inst, const OperandVector &Operands);
1788 void cvtSdwaVOP2(MCInst &Inst, const OperandVector &Operands);
1789 void cvtSdwaVOP2b(MCInst &Inst, const OperandVector &Operands);
1790 void cvtSdwaVOP2e(MCInst &Inst, const OperandVector &Operands);
1791 void cvtSdwaVOPC(MCInst &Inst, const OperandVector &Operands);
1792
1793 enum class SDWAInstType : unsigned { VOP1 = 0, VOP2 = 1, VOPC = 2 };
1794
1795 void cvtSDWA(MCInst &Inst, const OperandVector &Operands,
1796 SDWAInstType BasicInstType, bool SkipDstVcc = false,
1797 bool SkipSrcVcc = false);
1798
1799 ParseStatus parseEndpgm(OperandVector &Operands);
1800
1801 ParseStatus parseVOPD(OperandVector &Operands);
1802};
1803
1804} // end anonymous namespace
1805
1806// May be called with integer type with equivalent bitwidth.
1807static const fltSemantics *getFltSemantics(unsigned Size) {
1808 switch (Size) {
1809 case 4:
1810 return &APFloat::IEEEsingle();
1811 case 8:
1812 return &APFloat::IEEEdouble();
1813 case 2:
1814 return &APFloat::IEEEhalf();
1815 default:
1816 llvm_unreachable("unsupported fp type");
1817 }
1818}
1819
1821 return getFltSemantics(VT.getScalarSizeInBits() / 8);
1822}
1823
1825 switch (OperandType) {
1826 // When floating-point immediate is used as operand of type i16, the 32-bit
1827 // representation of the constant truncated to the 16 LSBs should be used.
1842 return &APFloat::IEEEsingle();
1851 return &APFloat::IEEEdouble();
1860 return &APFloat::IEEEhalf();
1865 return &APFloat::BFloat();
1866 default:
1867 llvm_unreachable("unsupported fp type");
1868 }
1869}
1870
1871//===----------------------------------------------------------------------===//
1872// Operand
1873//===----------------------------------------------------------------------===//
1874
1875static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT) {
1876 bool Lost;
1877
1878 // Convert literal to single precision
1879 APFloat::opStatus Status = FPLiteral.convert(
1881 // We allow precision lost but not overflow or underflow
1882 if (Status != APFloat::opOK && Lost &&
1883 ((Status & APFloat::opOverflow) != 0 ||
1884 (Status & APFloat::opUnderflow) != 0)) {
1885 return false;
1886 }
1887
1888 return true;
1889}
1890
1891static bool isSafeTruncation(int64_t Val, unsigned Size) {
1892 return isUIntN(Size, Val) || isIntN(Size, Val);
1893}
1894
1895static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi) {
1896 if (VT.getScalarType() == MVT::i16)
1897 return isInlinableLiteral32(Val, HasInv2Pi);
1898
1899 if (VT.getScalarType() == MVT::f16)
1900 return AMDGPU::isInlinableLiteralFP16(Val, HasInv2Pi);
1901
1902 assert(VT.getScalarType() == MVT::bf16);
1903
1904 return AMDGPU::isInlinableLiteralBF16(Val, HasInv2Pi);
1905}
1906
1907bool AMDGPUOperand::isInlinableImm(MVT type) const {
1908
1909 // This is a hack to enable named inline values like
1910 // shared_base with both 32-bit and 64-bit operands.
1911 // Note that these values are defined as
1912 // 32-bit operands only.
1913 if (isInlineValue()) {
1914 return true;
1915 }
1916
1917 if (!isImmTy(ImmTyNone)) {
1918 // Only plain immediates are inlinable (e.g. "clamp" attribute is not)
1919 return false;
1920 }
1921
1922 if (getModifiers().Lit != LitModifier::None)
1923 return false;
1924
1925 // TODO: We should avoid using host float here. It would be better to
1926 // check the float bit values which is what a few other places do.
1927 // We've had bot failures before due to weird NaN support on mips hosts.
1928
1929 APInt Literal(64, Imm.Val);
1930
1931 if (Imm.IsFPImm) { // We got fp literal token
1932 if (type == MVT::f64 || type == MVT::i64) { // Expected 64-bit operand
1934 AsmParser->hasInv2PiInlineImm());
1935 }
1936
1937 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64, Imm.Val));
1938 if (!canLosslesslyConvertToFPType(FPLiteral, type))
1939 return false;
1940
1941 if (type.getScalarSizeInBits() == 16) {
1942 bool Lost = false;
1943 switch (type.getScalarType().SimpleTy) {
1944 default:
1945 llvm_unreachable("unknown 16-bit type");
1946 case MVT::bf16:
1947 FPLiteral.convert(APFloatBase::BFloat(), APFloat::rmNearestTiesToEven,
1948 &Lost);
1949 break;
1950 case MVT::f16:
1951 FPLiteral.convert(APFloatBase::IEEEhalf(), APFloat::rmNearestTiesToEven,
1952 &Lost);
1953 break;
1954 case MVT::i16:
1955 FPLiteral.convert(APFloatBase::IEEEsingle(),
1956 APFloat::rmNearestTiesToEven, &Lost);
1957 break;
1958 }
1959 // We need to use 32-bit representation here because when a floating-point
1960 // inline constant is used as an i16 operand, its 32-bit representation
1961 // representation will be used. We will need the 32-bit value to check if
1962 // it is FP inline constant.
1963 uint32_t ImmVal = FPLiteral.bitcastToAPInt().getZExtValue();
1964 return isInlineableLiteralOp16(ImmVal, type,
1965 AsmParser->hasInv2PiInlineImm());
1966 }
1967
1968 // Check if single precision literal is inlinable
1970 static_cast<int32_t>(FPLiteral.bitcastToAPInt().getZExtValue()),
1971 AsmParser->hasInv2PiInlineImm());
1972 }
1973
1974 // We got int literal token.
1975 if (type == MVT::f64 || type == MVT::i64) { // Expected 64-bit operand
1977 AsmParser->hasInv2PiInlineImm());
1978 }
1979
1980 if (!isSafeTruncation(Imm.Val, type.getScalarSizeInBits())) {
1981 return false;
1982 }
1983
1984 if (type.getScalarSizeInBits() == 16) {
1986 static_cast<int16_t>(Literal.getLoBits(16).getSExtValue()), type,
1987 AsmParser->hasInv2PiInlineImm());
1988 }
1989
1991 static_cast<int32_t>(Literal.getLoBits(32).getZExtValue()),
1992 AsmParser->hasInv2PiInlineImm());
1993}
1994
1995bool AMDGPUOperand::isLiteralImm(MVT type) const {
1996 // Check that this immediate can be added as literal
1997 if (!isImmTy(ImmTyNone)) {
1998 return false;
1999 }
2000
2001 bool Allow64Bit =
2002 (type == MVT::i64 || type == MVT::f64) && AsmParser->has64BitLiterals();
2003
2004 if (!Imm.IsFPImm) {
2005 // We got int literal token.
2006
2007 if (type == MVT::f64 && hasFPModifiers()) {
2008 // Cannot apply fp modifiers to int literals preserving the same semantics
2009 // for VOP1/2/C and VOP3 because of integer truncation. To avoid
2010 // ambiguity, disable these cases.
2011 return false;
2012 }
2013
2014 unsigned Size = type.getSizeInBits();
2015 if (Size == 64) {
2016 if (Allow64Bit && !AMDGPU::isValid32BitLiteral(Imm.Val, false))
2017 return true;
2018 Size = 32;
2019 }
2020
2021 // FIXME: 64-bit operands can zero extend, sign extend, or pad zeroes for FP
2022 // types.
2023 return isSafeTruncation(Imm.Val, Size);
2024 }
2025
2026 // We got fp literal token
2027 if (type == MVT::f64) { // Expected 64-bit fp operand
2028 // We would set low 64-bits of literal to zeroes but we accept this literals
2029 return true;
2030 }
2031
2032 if (type == MVT::i64) { // Expected 64-bit int operand
2033 // We don't allow fp literals in 64-bit integer instructions. It is
2034 // unclear how we should encode them.
2035 return false;
2036 }
2037
2038 // We allow fp literals with f16x2 operands assuming that the specified
2039 // literal goes into the lower half and the upper half is zero. We also
2040 // require that the literal may be losslessly converted to f16.
2041 //
2042 // For i16x2 operands, we assume that the specified literal is encoded as a
2043 // single-precision float. This is pretty odd, but it matches SP3 and what
2044 // happens in hardware.
2045 MVT ExpectedType = (type == MVT::v2f16) ? MVT::f16
2046 : (type == MVT::v2i16) ? MVT::f32
2047 : (type == MVT::v2f32) ? MVT::f32
2048 : type;
2049
2050 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64, Imm.Val));
2051 return canLosslesslyConvertToFPType(FPLiteral, ExpectedType);
2052}
2053
2054bool AMDGPUOperand::isRegClassTarget(unsigned TargetRCIdx) const {
2055 if (!isRegKind())
2056 return false;
2057 int16_t RCID = AsmParser->getTargetRegClass(TargetRCIdx);
2058 return RCID >= 0 && isRegClass(RCID);
2059}
2060
2061bool AMDGPUOperand::isRegClass(unsigned RCID) const {
2062 return isRegKind() &&
2063 AsmParser->getMRI()->getRegClass(RCID).contains(getReg());
2064}
2065
2066bool AMDGPUOperand::isVRegWithInputMods() const {
2067 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
2068 // GFX90A allows DPP on 64-bit operands.
2069 (AsmParser->getFeatureBits()[AMDGPU::FeatureDPALU_DPP] &&
2070 isRegClassTarget(AMDGPU::VReg_64_AlignTarget));
2071}
2072
2073template <bool IsFake16>
2074bool AMDGPUOperand::isT16_Lo128VRegWithInputMods() const {
2075 return isRegClass(IsFake16 ? AMDGPU::VGPR_32_Lo128RegClassID
2076 : AMDGPU::VGPR_16_Lo128RegClassID);
2077}
2078
2079template <bool IsFake16> bool AMDGPUOperand::isT16VRegWithInputMods() const {
2080 return isRegClass(IsFake16 ? AMDGPU::VGPR_32RegClassID
2081 : AMDGPU::VGPR_16RegClassID);
2082}
2083
2084bool AMDGPUOperand::isT16_LO16VRegWithInputMods() const {
2085 return isRegClass(AMDGPU::VGPR_16_LO16RegClassID);
2086}
2087
2088bool AMDGPUOperand::isSDWAOperand(MVT type) const {
2089 if (AsmParser->isVI())
2090 return isVReg32();
2091 if (AsmParser->isGFX9Plus())
2092 return isRegClass(AMDGPU::VS_32RegClassID) || isInlinableImm(type);
2093 return false;
2094}
2095
2096bool AMDGPUOperand::isSDWAFP16Operand() const {
2097 return isSDWAOperand(MVT::f16);
2098}
2099
2100bool AMDGPUOperand::isSDWAFP32Operand() const {
2101 return isSDWAOperand(MVT::f32);
2102}
2103
2104bool AMDGPUOperand::isSDWAInt16Operand() const {
2105 return isSDWAOperand(MVT::i16);
2106}
2107
2108bool AMDGPUOperand::isSDWAInt32Operand() const {
2109 return isSDWAOperand(MVT::i32);
2110}
2111
2112bool AMDGPUOperand::isBoolReg() const {
2113 return isReg() && ((AsmParser->isWave64() && isSCSrc_b64()) ||
2114 (AsmParser->isWave32() && isSCSrc_b32()));
2115}
2116
2117uint64_t AMDGPUOperand::applyInputFPModifiers(uint64_t Val,
2118 unsigned Size) const {
2119 assert(isImmTy(ImmTyNone) && Imm.Mods.hasFPModifiers());
2120 assert(Size == 2 || Size == 4 || Size == 8);
2121
2122 const uint64_t FpSignMask = (1ULL << (Size * 8 - 1));
2123
2124 if (Imm.Mods.Abs) {
2125 Val &= ~FpSignMask;
2126 }
2127 if (Imm.Mods.Neg) {
2128 Val ^= FpSignMask;
2129 }
2130
2131 return Val;
2132}
2133
2134void AMDGPUOperand::addImmOperands(MCInst &Inst, unsigned N,
2135 bool ApplyModifiers) const {
2136 MCOpIdx = Inst.getNumOperands();
2137
2138 if (isExpr()) {
2140 return;
2141 }
2142
2143 if (AMDGPU::isSISrcOperand(AsmParser->getMII()->get(Inst.getOpcode()),
2144 Inst.getNumOperands())) {
2145 addLiteralImmOperand(Inst, Imm.Val,
2146 ApplyModifiers & isImmTy(ImmTyNone) &&
2147 Imm.Mods.hasFPModifiers());
2148 } else {
2149 assert(!isImmTy(ImmTyNone) || !hasModifiers());
2151 }
2152}
2153
2154void AMDGPUOperand::addLiteralImmOperand(MCInst &Inst, int64_t Val,
2155 bool ApplyModifiers) const {
2156 const auto &InstDesc = AsmParser->getMII()->get(Inst.getOpcode());
2157 auto OpNum = Inst.getNumOperands();
2158 // Check that this operand accepts literals
2159 assert(AMDGPU::isSISrcOperand(InstDesc, OpNum));
2160
2161 if (ApplyModifiers) {
2162 assert(AMDGPU::isSISrcFPOperand(InstDesc, OpNum));
2163 const unsigned Size =
2164 Imm.IsFPImm ? sizeof(double) : getOperandSize(InstDesc, OpNum);
2165 Val = applyInputFPModifiers(Val, Size);
2166 }
2167
2168 APInt Literal(64, Val);
2169 uint8_t OpTy = InstDesc.operands()[OpNum].OperandType;
2170
2171 bool CanUse64BitLiterals =
2172 AsmParser->has64BitLiterals() && !SIInstrFlags::isVOP3Like(InstDesc);
2173 LitModifier Lit = getModifiers().Lit;
2174 MCContext &Ctx = AsmParser->getContext();
2175
2176 if (Imm.IsFPImm) { // We got fp literal token
2177 switch (OpTy) {
2185 if (Lit == LitModifier::None &&
2187 AsmParser->hasInv2PiInlineImm())) {
2188 Inst.addOperand(MCOperand::createImm(Literal.getZExtValue()));
2189 return;
2190 }
2191
2192 // Non-inlineable
2193 if (AMDGPU::isSISrcFPOperand(InstDesc,
2194 OpNum)) { // Expected 64-bit fp operand
2195 bool HasMandatoryLiteral =
2196 AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::imm);
2197 // For fp operands we check if low 32 bits are zeros
2198 if (Literal.getLoBits(32) != 0 &&
2199 (InstDesc.getSize() != 4 || !AsmParser->has64BitLiterals()) &&
2200 !HasMandatoryLiteral) {
2201 const_cast<AMDGPUAsmParser *>(AsmParser)->Warning(
2202 Inst.getLoc(),
2203 "Can't encode literal as exact 64-bit floating-point operand. "
2204 "Low 32-bits will be set to zero");
2205 Val &= 0xffffffff00000000u;
2206 }
2207
2208 if ((OpTy == AMDGPU::OPERAND_REG_IMM_FP64 ||
2211 if (CanUse64BitLiterals && Lit == LitModifier::None &&
2212 (isInt<32>(Val) || isUInt<32>(Val))) {
2213 // The floating-point operand will be verbalized as an
2214 // integer one. If that integer happens to fit 32 bits, on
2215 // re-assembling it will be intepreted as the high half of
2216 // the actual value, so we have to wrap it into lit64().
2217 Lit = LitModifier::Lit64;
2218 } else if (Lit == LitModifier::Lit) {
2219 // For FP64 operands lit() specifies the high half of the value.
2220 Val = Hi_32(Val);
2221 }
2222 }
2223 break;
2224 }
2225
2226 // We don't allow fp literals in 64-bit integer instructions. It is
2227 // unclear how we should encode them. This case should be checked earlier
2228 // in predicate methods (isLiteralImm())
2229 llvm_unreachable("fp literal in 64-bit integer instruction.");
2230
2232 if (CanUse64BitLiterals && Lit == LitModifier::None &&
2233 (isInt<32>(Val) || isUInt<32>(Val)))
2234 Lit = LitModifier::Lit64;
2235 break;
2236
2241 if (Lit == LitModifier::None && AsmParser->hasInv2PiInlineImm() &&
2242 Literal == 0x3fc45f306725feed) {
2243 // This is the 1/(2*pi) which is going to be truncated to bf16 with the
2244 // loss of precision. The constant represents ideomatic fp32 value of
2245 // 1/(2*pi) = 0.15915494 since bf16 is in fact fp32 with cleared low 16
2246 // bits. Prevent rounding below.
2247 Inst.addOperand(MCOperand::createImm(0x3e22));
2248 return;
2249 }
2250 [[fallthrough]];
2251
2274 bool lost;
2275 APFloat FPLiteral(APFloat::IEEEdouble(), Literal);
2276 // Convert literal to single precision
2277 FPLiteral.convert(*getOpFltSemantics(OpTy), APFloat::rmNearestTiesToEven,
2278 &lost);
2279 // We allow precision lost but not overflow or underflow. This should be
2280 // checked earlier in isLiteralImm()
2281
2282 Val = FPLiteral.bitcastToAPInt().getZExtValue();
2283 break;
2284 }
2285 default:
2286 llvm_unreachable("invalid operand size");
2287 }
2288
2289 if (Lit != LitModifier::None) {
2290 Inst.addOperand(
2292 } else {
2294 }
2295 return;
2296 }
2297
2298 // We got int literal token.
2299 // Only sign extend inline immediates.
2300 switch (OpTy) {
2315 break;
2316
2320 if (Lit == LitModifier::None &&
2321 AMDGPU::isInlinableLiteral64(Val, AsmParser->hasInv2PiInlineImm())) {
2323 return;
2324 }
2325
2326 // When the 32 MSBs are not zero (effectively means it can't be safely
2327 // truncated to uint32_t), if the target doesn't support 64-bit literals, or
2328 // the lit modifier is explicitly used, we need to truncate it to the 32
2329 // LSBs.
2330 if (!AsmParser->has64BitLiterals() || Lit == LitModifier::Lit)
2331 Val = Lo_32(Val);
2332 break;
2333
2338 if (Lit == LitModifier::None &&
2339 AMDGPU::isInlinableLiteral64(Val, AsmParser->hasInv2PiInlineImm())) {
2341 return;
2342 }
2343
2344 // If the target doesn't support 64-bit literals, we need to use the
2345 // constant as the high 32 MSBs of a double-precision floating point value.
2346 if (!AsmParser->has64BitLiterals()) {
2347 Val = static_cast<uint64_t>(Val) << 32;
2348 } else {
2349 // Now the target does support 64-bit literals, there are two cases
2350 // where we still want to use src_literal encoding:
2351 // 1) explicitly forced by using lit modifier;
2352 // 2) the value is a valid 32-bit representation (signed or unsigned),
2353 // meanwhile not forced by lit64 modifier.
2354 if (Lit == LitModifier::Lit ||
2355 (Lit != LitModifier::Lit64 && (isInt<32>(Val) || isUInt<32>(Val))))
2356 Val = static_cast<uint64_t>(Val) << 32;
2357 }
2358
2359 // For FP64 operands lit() specifies the high half of the value.
2360 if (Lit == LitModifier::Lit)
2361 Val = Hi_32(Val);
2362 break;
2363
2376 break;
2377
2379 if ((isInt<32>(Val) || isUInt<32>(Val)) && Lit != LitModifier::Lit64)
2380 Val <<= 32;
2381 break;
2382
2383 default:
2384 llvm_unreachable("invalid operand type");
2385 }
2386
2387 if (Lit != LitModifier::None) {
2388 Inst.addOperand(
2390 } else {
2392 }
2393}
2394
2395void AMDGPUOperand::addRegOperands(MCInst &Inst, unsigned N) const {
2396 MCOpIdx = Inst.getNumOperands();
2397 Inst.addOperand(
2398 MCOperand::createReg(AMDGPU::getMCReg(getReg(), AsmParser->getSTI())));
2399}
2400
2401bool AMDGPUOperand::isInlineValue() const {
2402 return isRegKind() && ::isInlineValue(getReg());
2403}
2404
2405//===----------------------------------------------------------------------===//
2406// AsmParser
2407//===----------------------------------------------------------------------===//
2408
2409void AMDGPUAsmParser::createConstantSymbol(StringRef Id, int64_t Val) {
2410 // TODO: make those pre-defined variables read-only.
2411 // Currently there is none suitable machinery in the core llvm-mc for this.
2412 // MCSymbol::isRedefinable is intended for another purpose, and
2413 // AsmParser::parseDirectiveSet() cannot be specialized for specific target.
2414 MCContext &Ctx = getContext();
2415 MCSymbol *Sym = Ctx.getOrCreateSymbol(Id);
2417}
2418
2419static int getRegClass(RegisterKind Is, unsigned RegWidth) {
2420 if (Is == IS_VGPR) {
2421 switch (RegWidth) {
2422 default:
2423 return -1;
2424 case 32:
2425 return AMDGPU::VGPR_32RegClassID;
2426 case 64:
2427 return AMDGPU::VReg_64RegClassID;
2428 case 96:
2429 return AMDGPU::VReg_96RegClassID;
2430 case 128:
2431 return AMDGPU::VReg_128RegClassID;
2432 case 160:
2433 return AMDGPU::VReg_160RegClassID;
2434 case 192:
2435 return AMDGPU::VReg_192RegClassID;
2436 case 224:
2437 return AMDGPU::VReg_224RegClassID;
2438 case 256:
2439 return AMDGPU::VReg_256RegClassID;
2440 case 288:
2441 return AMDGPU::VReg_288RegClassID;
2442 case 320:
2443 return AMDGPU::VReg_320RegClassID;
2444 case 352:
2445 return AMDGPU::VReg_352RegClassID;
2446 case 384:
2447 return AMDGPU::VReg_384RegClassID;
2448 case 512:
2449 return AMDGPU::VReg_512RegClassID;
2450 case 1024:
2451 return AMDGPU::VReg_1024RegClassID;
2452 }
2453 } else if (Is == IS_TTMP) {
2454 switch (RegWidth) {
2455 default:
2456 return -1;
2457 case 32:
2458 return AMDGPU::TTMP_32RegClassID;
2459 case 64:
2460 return AMDGPU::TTMP_64RegClassID;
2461 case 128:
2462 return AMDGPU::TTMP_128RegClassID;
2463 case 256:
2464 return AMDGPU::TTMP_256RegClassID;
2465 case 512:
2466 return AMDGPU::TTMP_512RegClassID;
2467 }
2468 } else if (Is == IS_SGPR) {
2469 switch (RegWidth) {
2470 default:
2471 return -1;
2472 case 32:
2473 return AMDGPU::SGPR_32RegClassID;
2474 case 64:
2475 return AMDGPU::SGPR_64RegClassID;
2476 case 96:
2477 return AMDGPU::SGPR_96RegClassID;
2478 case 128:
2479 return AMDGPU::SGPR_128RegClassID;
2480 case 160:
2481 return AMDGPU::SGPR_160RegClassID;
2482 case 192:
2483 return AMDGPU::SGPR_192RegClassID;
2484 case 224:
2485 return AMDGPU::SGPR_224RegClassID;
2486 case 256:
2487 return AMDGPU::SGPR_256RegClassID;
2488 case 288:
2489 return AMDGPU::SGPR_288RegClassID;
2490 case 320:
2491 return AMDGPU::SGPR_320RegClassID;
2492 case 352:
2493 return AMDGPU::SGPR_352RegClassID;
2494 case 384:
2495 return AMDGPU::SGPR_384RegClassID;
2496 case 512:
2497 return AMDGPU::SGPR_512RegClassID;
2498 }
2499 } else if (Is == IS_AGPR) {
2500 switch (RegWidth) {
2501 default:
2502 return -1;
2503 case 32:
2504 return AMDGPU::AGPR_32RegClassID;
2505 case 64:
2506 return AMDGPU::AReg_64RegClassID;
2507 case 96:
2508 return AMDGPU::AReg_96RegClassID;
2509 case 128:
2510 return AMDGPU::AReg_128RegClassID;
2511 case 160:
2512 return AMDGPU::AReg_160RegClassID;
2513 case 192:
2514 return AMDGPU::AReg_192RegClassID;
2515 case 224:
2516 return AMDGPU::AReg_224RegClassID;
2517 case 256:
2518 return AMDGPU::AReg_256RegClassID;
2519 case 288:
2520 return AMDGPU::AReg_288RegClassID;
2521 case 320:
2522 return AMDGPU::AReg_320RegClassID;
2523 case 352:
2524 return AMDGPU::AReg_352RegClassID;
2525 case 384:
2526 return AMDGPU::AReg_384RegClassID;
2527 case 512:
2528 return AMDGPU::AReg_512RegClassID;
2529 case 1024:
2530 return AMDGPU::AReg_1024RegClassID;
2531 }
2532 }
2533 return -1;
2534}
2535
2538 .Case("exec", AMDGPU::EXEC)
2539 .Case("vcc", AMDGPU::VCC)
2540 .Case("flat_scratch", AMDGPU::FLAT_SCR)
2541 .Case("xnack_mask", AMDGPU::XNACK_MASK)
2542 .Case("shared_base", AMDGPU::SRC_SHARED_BASE)
2543 .Case("src_shared_base", AMDGPU::SRC_SHARED_BASE)
2544 .Case("shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2545 .Case("src_shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2546 .Case("private_base", AMDGPU::SRC_PRIVATE_BASE)
2547 .Case("src_private_base", AMDGPU::SRC_PRIVATE_BASE)
2548 .Case("private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2549 .Case("src_private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2550 .Case("src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
2551 .Case("src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
2552 .Case("pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2553 .Case("src_pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2554 .Case("lds_direct", AMDGPU::LDS_DIRECT)
2555 .Case("src_lds_direct", AMDGPU::LDS_DIRECT)
2556 .Case("m0", AMDGPU::M0)
2557 .Case("vccz", AMDGPU::SRC_VCCZ)
2558 .Case("src_vccz", AMDGPU::SRC_VCCZ)
2559 .Case("execz", AMDGPU::SRC_EXECZ)
2560 .Case("src_execz", AMDGPU::SRC_EXECZ)
2561 .Case("scc", AMDGPU::SRC_SCC)
2562 .Case("src_scc", AMDGPU::SRC_SCC)
2563 .Case("tba", AMDGPU::TBA)
2564 .Case("tma", AMDGPU::TMA)
2565 .Case("flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
2566 .Case("flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
2567 .Case("xnack_mask_lo", AMDGPU::XNACK_MASK_LO)
2568 .Case("xnack_mask_hi", AMDGPU::XNACK_MASK_HI)
2569 .Case("vcc_lo", AMDGPU::VCC_LO)
2570 .Case("vcc_hi", AMDGPU::VCC_HI)
2571 .Case("exec_lo", AMDGPU::EXEC_LO)
2572 .Case("exec_hi", AMDGPU::EXEC_HI)
2573 .Case("tma_lo", AMDGPU::TMA_LO)
2574 .Case("tma_hi", AMDGPU::TMA_HI)
2575 .Case("tba_lo", AMDGPU::TBA_LO)
2576 .Case("tba_hi", AMDGPU::TBA_HI)
2577 .Case("pc", AMDGPU::PC_REG)
2578 .Case("null", AMDGPU::SGPR_NULL)
2579 .Default(AMDGPU::NoRegister);
2580}
2581
2582bool AMDGPUAsmParser::ParseRegister(MCRegister &RegNo, SMLoc &StartLoc,
2583 SMLoc &EndLoc, bool RestoreOnFailure) {
2584 auto R = parseRegister();
2585 if (!R)
2586 return true;
2587 assert(R->isReg());
2588 RegNo = R->getReg();
2589 StartLoc = R->getStartLoc();
2590 EndLoc = R->getEndLoc();
2591 return false;
2592}
2593
2594bool AMDGPUAsmParser::parseRegister(MCRegister &Reg, SMLoc &StartLoc,
2595 SMLoc &EndLoc) {
2596 return ParseRegister(Reg, StartLoc, EndLoc, /*RestoreOnFailure=*/false);
2597}
2598
2599ParseStatus AMDGPUAsmParser::tryParseRegister(MCRegister &Reg, SMLoc &StartLoc,
2600 SMLoc &EndLoc) {
2601 bool Result = ParseRegister(Reg, StartLoc, EndLoc, /*RestoreOnFailure=*/true);
2602 bool PendingErrors = getParser().hasPendingError();
2603 getParser().clearPendingErrors();
2604 if (PendingErrors)
2605 return ParseStatus::Failure;
2606 if (Result)
2607 return ParseStatus::NoMatch;
2608 return ParseStatus::Success;
2609}
2610
2611bool AMDGPUAsmParser::AddNextRegisterToList(MCRegister &Reg, unsigned &RegWidth,
2612 RegisterKind RegKind,
2613 MCRegister Reg1,
2614 RegisterKind RegKind1, SMLoc Loc) {
2615 // Allow VCC_LO/HI at the end of SGPR lists.
2616 if (RegKind == IS_SGPR) {
2617 unsigned RegIdx = (Reg - AMDGPU::SGPR0) + RegWidth / 32;
2618 if ((RegIdx == 106 && Reg1 == AMDGPU::VCC_LO) ||
2619 (RegIdx == 107 && Reg1 == AMDGPU::VCC_HI)) {
2620 RegWidth += 32;
2621 return true;
2622 }
2623 }
2624
2625 if (RegKind != RegKind1) {
2626 Error(Loc, "registers in a list must be of the same kind");
2627 return false;
2628 }
2629
2630 switch (RegKind) {
2631 case IS_SPECIAL:
2632 if (Reg == AMDGPU::EXEC_LO && Reg1 == AMDGPU::EXEC_HI) {
2633 Reg = AMDGPU::EXEC;
2634 RegWidth = 64;
2635 return true;
2636 }
2637 if (Reg == AMDGPU::FLAT_SCR_LO && Reg1 == AMDGPU::FLAT_SCR_HI) {
2638 Reg = AMDGPU::FLAT_SCR;
2639 RegWidth = 64;
2640 return true;
2641 }
2642 if (Reg == AMDGPU::XNACK_MASK_LO && Reg1 == AMDGPU::XNACK_MASK_HI) {
2643 Reg = AMDGPU::XNACK_MASK;
2644 RegWidth = 64;
2645 return true;
2646 }
2647 if (Reg == AMDGPU::VCC_LO && Reg1 == AMDGPU::VCC_HI) {
2648 Reg = AMDGPU::VCC;
2649 RegWidth = 64;
2650 return true;
2651 }
2652 if (Reg == AMDGPU::TBA_LO && Reg1 == AMDGPU::TBA_HI) {
2653 Reg = AMDGPU::TBA;
2654 RegWidth = 64;
2655 return true;
2656 }
2657 if (Reg == AMDGPU::TMA_LO && Reg1 == AMDGPU::TMA_HI) {
2658 Reg = AMDGPU::TMA;
2659 RegWidth = 64;
2660 return true;
2661 }
2662 Error(Loc, "register does not fit in the list");
2663 return false;
2664 case IS_VGPR:
2665 case IS_SGPR:
2666 case IS_AGPR:
2667 case IS_TTMP:
2668 if (Reg1 != Reg + RegWidth / 32) {
2669 Error(Loc, "registers in a list must have consecutive indices");
2670 return false;
2671 }
2672 RegWidth += 32;
2673 return true;
2674 default:
2675 llvm_unreachable("unexpected register kind");
2676 }
2677}
2678
2679struct RegInfo {
2681 RegisterKind Kind;
2682};
2683
2684static constexpr RegInfo RegularRegisters[] = {
2685 {{"v"}, IS_VGPR}, {{"s"}, IS_SGPR}, {{"ttmp"}, IS_TTMP},
2686 {{"acc"}, IS_AGPR}, {{"a"}, IS_AGPR},
2687};
2688
2689static bool isRegularReg(RegisterKind Kind) {
2690 return Kind == IS_VGPR || Kind == IS_SGPR || Kind == IS_TTMP ||
2691 Kind == IS_AGPR;
2692}
2693
2695 for (const RegInfo &Reg : RegularRegisters)
2696 if (Str.starts_with(Reg.Name))
2697 return &Reg;
2698 return nullptr;
2699}
2700
2701static bool getRegNum(StringRef Str, unsigned &Num) {
2702 return !Str.getAsInteger(10, Num);
2703}
2704
2705bool AMDGPUAsmParser::isRegister(const AsmToken &Token,
2706 const AsmToken &NextToken) const {
2707
2708 // A list of consecutive registers: [s0,s1,s2,s3]
2709 if (Token.is(AsmToken::LBrac))
2710 return true;
2711
2712 if (!Token.is(AsmToken::Identifier))
2713 return false;
2714
2715 // A single register like s0 or a range of registers like s[0:1]
2716
2717 StringRef Str = Token.getString();
2718 const RegInfo *Reg = getRegularRegInfo(Str);
2719 if (Reg) {
2720 StringRef RegName = Reg->Name;
2721 StringRef RegSuffix = Str.substr(RegName.size());
2722 if (!RegSuffix.empty()) {
2723 RegSuffix.consume_back(".l");
2724 RegSuffix.consume_back(".h");
2725 unsigned Num;
2726 // A single register with an index: rXX
2727 if (getRegNum(RegSuffix, Num))
2728 return true;
2729 } else {
2730 // A range of registers: r[XX:YY].
2731 if (NextToken.is(AsmToken::LBrac))
2732 return true;
2733 }
2734 }
2735
2736 return getSpecialRegForName(Str).isValid();
2737}
2738
2739bool AMDGPUAsmParser::isRegister() {
2740 return isRegister(getToken(), peekToken());
2741}
2742
2743MCRegister AMDGPUAsmParser::getRegularReg(RegisterKind RegKind, unsigned RegNum,
2744 unsigned SubReg, unsigned RegWidth,
2745 SMLoc Loc) {
2746 assert(isRegularReg(RegKind));
2747
2748 unsigned AlignSize = 1;
2749 if (RegKind == IS_SGPR || RegKind == IS_TTMP) {
2750 // SGPR and TTMP registers must be aligned.
2751 // Max required alignment is 4 dwords.
2752 AlignSize = std::min(llvm::bit_ceil(RegWidth / 32), 4u);
2753 }
2754
2755 if (RegNum % AlignSize != 0) {
2756 Error(Loc, "invalid register alignment");
2757 return MCRegister();
2758 }
2759
2760 unsigned RegIdx = RegNum / AlignSize;
2761 int RCID = getRegClass(RegKind, RegWidth);
2762 if (RCID == -1) {
2763 Error(Loc, "invalid or unsupported register size");
2764 return MCRegister();
2765 }
2766
2767 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
2768 const MCRegisterClass &RC = TRI->getRegClass(RCID);
2769 if (RegIdx >= RC.getNumRegs() || (RegKind == IS_VGPR && RegIdx > 255)) {
2770 Error(Loc, "register index is out of range");
2771 return AMDGPU::NoRegister;
2772 }
2773
2774 if (RegKind == IS_VGPR && !isGFX1250Plus() && RegIdx + RegWidth / 32 > 256) {
2775 Error(Loc, "register index is out of range");
2776 return MCRegister();
2777 }
2778
2779 MCRegister Reg = RC.getRegister(RegIdx);
2780
2781 if (SubReg) {
2782 Reg = TRI->getSubReg(Reg, SubReg);
2783
2784 if (!Reg)
2785 Error(Loc, "invalid subregister");
2786 }
2787
2788 return Reg;
2789}
2790
2791bool AMDGPUAsmParser::ParseRegRange(unsigned &Num, unsigned &RegWidth,
2792 unsigned &SubReg) {
2793 int64_t RegLo, RegHi;
2794 if (!skipToken(AsmToken::LBrac, "missing register index"))
2795 return false;
2796
2797 SMLoc FirstIdxLoc = getLoc();
2798 SMLoc SecondIdxLoc;
2799
2800 if (!parseExpr(RegLo))
2801 return false;
2802
2803 if (trySkipToken(AsmToken::Colon)) {
2804 SecondIdxLoc = getLoc();
2805 if (!parseExpr(RegHi))
2806 return false;
2807 } else {
2808 RegHi = RegLo;
2809 }
2810
2811 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
2812 return false;
2813
2814 if (!isUInt<32>(RegLo)) {
2815 Error(FirstIdxLoc, "invalid register index");
2816 return false;
2817 }
2818
2819 if (!isUInt<32>(RegHi)) {
2820 Error(SecondIdxLoc, "invalid register index");
2821 return false;
2822 }
2823
2824 if (RegLo > RegHi) {
2825 Error(FirstIdxLoc, "first register index should not exceed second index");
2826 return false;
2827 }
2828
2829 if (RegHi == RegLo) {
2830 StringRef RegSuffix = getTokenStr();
2831 if (RegSuffix == ".l") {
2832 SubReg = AMDGPU::lo16;
2833 lex();
2834 } else if (RegSuffix == ".h") {
2835 SubReg = AMDGPU::hi16;
2836 lex();
2837 }
2838 }
2839
2840 Num = static_cast<unsigned>(RegLo);
2841 RegWidth = 32 * ((RegHi - RegLo) + 1);
2842
2843 return true;
2844}
2845
2846MCRegister AMDGPUAsmParser::ParseSpecialReg(RegisterKind &RegKind,
2847 unsigned &RegNum,
2848 unsigned &RegWidth,
2849 SmallVectorImpl<AsmToken> &Tokens) {
2850 assert(isToken(AsmToken::Identifier));
2851 MCRegister Reg = getSpecialRegForName(getTokenStr());
2852 if (Reg) {
2853 RegNum = 0;
2854 RegWidth = 32;
2855 RegKind = IS_SPECIAL;
2856 Tokens.push_back(getToken());
2857 lex(); // skip register name
2858 }
2859 return Reg;
2860}
2861
2862MCRegister AMDGPUAsmParser::ParseRegularReg(RegisterKind &RegKind,
2863 unsigned &RegNum,
2864 unsigned &RegWidth,
2865 SmallVectorImpl<AsmToken> &Tokens) {
2866 assert(isToken(AsmToken::Identifier));
2867 StringRef RegName = getTokenStr();
2868 auto Loc = getLoc();
2869
2870 const RegInfo *RI = getRegularRegInfo(RegName);
2871 if (!RI) {
2872 Error(Loc, "invalid register name");
2873 return MCRegister();
2874 }
2875
2876 Tokens.push_back(getToken());
2877 lex(); // skip register name
2878
2879 RegKind = RI->Kind;
2880 StringRef RegSuffix = RegName.substr(RI->Name.size());
2881 unsigned SubReg = NoSubRegister;
2882 bool IsRange = false;
2883 if (!RegSuffix.empty()) {
2884 if (RegSuffix.consume_back(".l"))
2885 SubReg = AMDGPU::lo16;
2886 else if (RegSuffix.consume_back(".h"))
2887 SubReg = AMDGPU::hi16;
2888
2889 // Single 32-bit register: vXX.
2890 if (!getRegNum(RegSuffix, RegNum)) {
2891 Error(Loc, "invalid register index");
2892 return MCRegister();
2893 }
2894 RegWidth = 32;
2895 } else {
2896 // Range of registers: v[XX:YY]. ":YY" is optional.
2897 IsRange = true;
2898 if (!ParseRegRange(RegNum, RegWidth, SubReg))
2899 return MCRegister();
2900 }
2901
2902 // Do not allow vcc_lo/hi be referred as s106/107.
2903 MCRegister Reg = getRegularReg(RegKind, RegNum, SubReg, RegWidth, Loc);
2904 const MCRegisterInfo &TRI = *getContext().getRegisterInfo();
2905 if (RegKind == IS_SGPR && IsRange
2906 ? (TRI.isSubRegister(Reg, VCC_LO) || TRI.isSubRegister(Reg, VCC_HI))
2907 : (Reg == VCC_LO || Reg == VCC_HI)) {
2908 Error(Loc, "register index is out of range");
2909 return MCRegister();
2910 }
2911
2912 return Reg;
2913}
2914
2915MCRegister AMDGPUAsmParser::ParseRegList(RegisterKind &RegKind,
2916 unsigned &RegNum, unsigned &RegWidth,
2917 SmallVectorImpl<AsmToken> &Tokens) {
2918 MCRegister Reg;
2919 auto ListLoc = getLoc();
2920
2921 if (!skipToken(AsmToken::LBrac,
2922 "expected a register or a list of registers")) {
2923 return MCRegister();
2924 }
2925
2926 // List of consecutive registers, e.g.: [s0,s1,s2,s3]
2927
2928 auto Loc = getLoc();
2929 if (!ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth))
2930 return MCRegister();
2931 if (RegWidth != 32) {
2932 Error(Loc, "expected a single 32-bit register");
2933 return MCRegister();
2934 }
2935
2936 for (; trySkipToken(AsmToken::Comma);) {
2937 RegisterKind NextRegKind;
2938 MCRegister NextReg;
2939 unsigned NextRegNum, NextRegWidth;
2940 Loc = getLoc();
2941
2942 if (!ParseAMDGPURegister(NextRegKind, NextReg, NextRegNum, NextRegWidth,
2943 Tokens)) {
2944 return MCRegister();
2945 }
2946 if (NextRegWidth != 32) {
2947 Error(Loc, "expected a single 32-bit register");
2948 return MCRegister();
2949 }
2950 if (!AddNextRegisterToList(Reg, RegWidth, RegKind, NextReg, NextRegKind,
2951 Loc))
2952 return MCRegister();
2953 }
2954
2955 if (!skipToken(AsmToken::RBrac,
2956 "expected a comma or a closing square bracket")) {
2957 return MCRegister();
2958 }
2959
2960 if (isRegularReg(RegKind))
2961 Reg = getRegularReg(RegKind, RegNum, NoSubRegister, RegWidth, ListLoc);
2962
2963 return Reg;
2964}
2965
2966bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
2967 MCRegister &Reg, unsigned &RegNum,
2968 unsigned &RegWidth,
2969 SmallVectorImpl<AsmToken> &Tokens) {
2970 auto Loc = getLoc();
2971 Reg = MCRegister();
2972
2973 if (isToken(AsmToken::Identifier)) {
2974 Reg = ParseSpecialReg(RegKind, RegNum, RegWidth, Tokens);
2975 if (!Reg)
2976 Reg = ParseRegularReg(RegKind, RegNum, RegWidth, Tokens);
2977 } else {
2978 Reg = ParseRegList(RegKind, RegNum, RegWidth, Tokens);
2979 }
2980
2981 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
2982 if (!Reg) {
2983 assert(Parser.hasPendingError());
2984 return false;
2985 }
2986
2987 if (!subtargetHasRegister(*TRI, Reg)) {
2988 if (Reg == AMDGPU::SGPR_NULL) {
2989 Error(Loc, "'null' operand is not supported on this GPU");
2990 } else {
2992 " register not available on this GPU");
2993 }
2994 return false;
2995 }
2996
2997 return true;
2998}
2999
3000bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3001 MCRegister &Reg, unsigned &RegNum,
3002 unsigned &RegWidth,
3003 bool RestoreOnFailure /*=false*/) {
3004 Reg = MCRegister();
3005
3007 if (ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth, Tokens)) {
3008 if (RestoreOnFailure) {
3009 while (!Tokens.empty()) {
3010 getLexer().UnLex(Tokens.pop_back_val());
3011 }
3012 }
3013 return true;
3014 }
3015 return false;
3016}
3017
3018std::optional<StringRef>
3019AMDGPUAsmParser::getGprCountSymbolName(RegisterKind RegKind) {
3020 switch (RegKind) {
3021 case IS_VGPR:
3022 return StringRef(".amdgcn.next_free_vgpr");
3023 case IS_SGPR:
3024 return StringRef(".amdgcn.next_free_sgpr");
3025 default:
3026 return std::nullopt;
3027 }
3028}
3029
3030void AMDGPUAsmParser::initializeGprCountSymbol(RegisterKind RegKind) {
3031 auto SymbolName = getGprCountSymbolName(RegKind);
3032 assert(SymbolName && "initializing invalid register kind");
3033 MCSymbol *Sym = getContext().getOrCreateSymbol(*SymbolName);
3035 Sym->setRedefinable(true);
3036}
3037
3038bool AMDGPUAsmParser::updateGprCountSymbols(RegisterKind RegKind,
3039 unsigned DwordRegIndex,
3040 unsigned RegWidth) {
3041 // Symbols are only defined for GCN targets
3042 if (ISA.Major < 6)
3043 return true;
3044
3045 auto SymbolName = getGprCountSymbolName(RegKind);
3046 if (!SymbolName)
3047 return true;
3048 MCSymbol *Sym = getContext().getOrCreateSymbol(*SymbolName);
3049
3050 int64_t NewMax = DwordRegIndex + divideCeil(RegWidth, 32) - 1;
3051 int64_t OldCount;
3052
3053 if (!Sym->isVariable())
3054 return !Error(getLoc(),
3055 ".amdgcn.next_free_{v,s}gpr symbols must be variable");
3056 if (!Sym->getVariableValue()->evaluateAsAbsolute(OldCount))
3057 return !Error(
3058 getLoc(),
3059 ".amdgcn.next_free_{v,s}gpr symbols must be absolute expressions");
3060
3061 if (OldCount <= NewMax)
3063
3064 return true;
3065}
3066
3067std::unique_ptr<AMDGPUOperand>
3068AMDGPUAsmParser::parseRegister(bool RestoreOnFailure) {
3069 const auto &Tok = getToken();
3070 SMLoc StartLoc = Tok.getLoc();
3071 SMLoc EndLoc = Tok.getEndLoc();
3072 RegisterKind RegKind;
3073 MCRegister Reg;
3074 unsigned RegNum, RegWidth;
3075
3076 if (!ParseAMDGPURegister(RegKind, Reg, RegNum, RegWidth)) {
3077 return nullptr;
3078 }
3079 if (isHsaAbi(getSTI())) {
3080 if (!updateGprCountSymbols(RegKind, RegNum, RegWidth))
3081 return nullptr;
3082 } else
3083 KernelScope.usesRegister(RegKind, RegNum, RegWidth);
3084 return AMDGPUOperand::CreateReg(this, Reg, StartLoc, EndLoc);
3085}
3086
3087ParseStatus AMDGPUAsmParser::parseImm(OperandVector &Operands,
3088 bool HasSP3AbsModifier, LitModifier Lit) {
3089 // TODO: add syntactic sugar for 1/(2*PI)
3090
3091 if (isRegister() || isModifier())
3092 return ParseStatus::NoMatch;
3093
3094 if (Lit == LitModifier::None) {
3095 if (trySkipId("lit"))
3096 Lit = LitModifier::Lit;
3097 else if (trySkipId("lit64"))
3098 Lit = LitModifier::Lit64;
3099
3100 if (Lit != LitModifier::None) {
3101 if (!skipToken(AsmToken::LParen, "expected left paren after lit"))
3102 return ParseStatus::Failure;
3103 ParseStatus S = parseImm(Operands, HasSP3AbsModifier, Lit);
3104 if (S.isSuccess() &&
3105 !skipToken(AsmToken::RParen, "expected closing parentheses"))
3106 return ParseStatus::Failure;
3107 return S;
3108 }
3109 }
3110
3111 const auto &Tok = getToken();
3112 const auto &NextTok = peekToken();
3113 bool IsReal = Tok.is(AsmToken::Real);
3114 SMLoc S = getLoc();
3115 bool Negate = false;
3116
3117 if (!IsReal && Tok.is(AsmToken::Minus) && NextTok.is(AsmToken::Real)) {
3118 lex();
3119 IsReal = true;
3120 Negate = true;
3121 }
3122
3123 AMDGPUOperand::Modifiers Mods;
3124 Mods.Lit = Lit;
3125
3126 if (IsReal) {
3127 // Floating-point expressions are not supported.
3128 // Can only allow floating-point literals with an
3129 // optional sign.
3130
3131 StringRef Num = getTokenStr();
3132 lex();
3133
3134 APFloat RealVal(APFloat::IEEEdouble());
3135 auto roundMode = APFloat::rmNearestTiesToEven;
3136 if (errorToBool(RealVal.convertFromString(Num, roundMode).takeError()))
3137 return ParseStatus::Failure;
3138 if (Negate)
3139 RealVal.changeSign();
3140
3141 Operands.push_back(
3142 AMDGPUOperand::CreateImm(this, RealVal.bitcastToAPInt().getZExtValue(),
3143 S, AMDGPUOperand::ImmTyNone, true));
3144 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3145 Op.setModifiers(Mods);
3146
3147 return ParseStatus::Success;
3148
3149 } else {
3150 int64_t IntVal;
3151 const MCExpr *Expr;
3152 SMLoc S = getLoc();
3153
3154 if (HasSP3AbsModifier) {
3155 // This is a workaround for handling expressions
3156 // as arguments of SP3 'abs' modifier, for example:
3157 // |1.0|
3158 // |-1|
3159 // |1+x|
3160 // This syntax is not compatible with syntax of standard
3161 // MC expressions (due to the trailing '|').
3162 SMLoc EndLoc;
3163 if (getParser().parsePrimaryExpr(Expr, EndLoc, nullptr))
3164 return ParseStatus::Failure;
3165 } else {
3166 if (Parser.parseExpression(Expr))
3167 return ParseStatus::Failure;
3168 }
3169
3170 if (Expr->evaluateAsAbsolute(IntVal)) {
3171 if (Lit == LitModifier::Lit && !isInt<32>(IntVal) && !isUInt<32>(IntVal))
3172 return Error(S, "literal value out of range");
3173 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S));
3174 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3175 Op.setModifiers(Mods);
3176 } else {
3177 if (Lit != LitModifier::None)
3178 return ParseStatus::NoMatch;
3179 Operands.push_back(AMDGPUOperand::CreateExpr(this, Expr, S));
3180 }
3181
3182 return ParseStatus::Success;
3183 }
3184
3185 return ParseStatus::NoMatch;
3186}
3187
3188ParseStatus AMDGPUAsmParser::parseReg(OperandVector &Operands) {
3189 if (!isRegister())
3190 return ParseStatus::NoMatch;
3191
3192 if (auto R = parseRegister()) {
3193 assert(R->isReg());
3194 Operands.push_back(std::move(R));
3195 return ParseStatus::Success;
3196 }
3197 return ParseStatus::Failure;
3198}
3199
3200ParseStatus AMDGPUAsmParser::parseRegOrImm(OperandVector &Operands,
3201 bool HasSP3AbsMod, LitModifier Lit) {
3202 ParseStatus Res = parseReg(Operands);
3203 if (!Res.isNoMatch())
3204 return Res;
3205 if (isModifier())
3206 return ParseStatus::NoMatch;
3207 return parseImm(Operands, HasSP3AbsMod, Lit);
3208}
3209
3210bool AMDGPUAsmParser::isNamedOperandModifier(const AsmToken &Token,
3211 const AsmToken &NextToken) const {
3212 if (Token.is(AsmToken::Identifier) && NextToken.is(AsmToken::LParen)) {
3213 const auto &str = Token.getString();
3214 return str == "abs" || str == "neg" || str == "sext";
3215 }
3216 return false;
3217}
3218
3219bool AMDGPUAsmParser::isOpcodeModifierWithVal(const AsmToken &Token,
3220 const AsmToken &NextToken) const {
3221 return Token.is(AsmToken::Identifier) && NextToken.is(AsmToken::Colon);
3222}
3223
3224bool AMDGPUAsmParser::isOperandModifier(const AsmToken &Token,
3225 const AsmToken &NextToken) const {
3226 return isNamedOperandModifier(Token, NextToken) || Token.is(AsmToken::Pipe);
3227}
3228
3229bool AMDGPUAsmParser::isRegOrOperandModifier(const AsmToken &Token,
3230 const AsmToken &NextToken) const {
3231 return isRegister(Token, NextToken) || isOperandModifier(Token, NextToken);
3232}
3233
3234// Check if this is an operand modifier or an opcode modifier
3235// which may look like an expression but it is not. We should
3236// avoid parsing these modifiers as expressions. Currently
3237// recognized sequences are:
3238// |...|
3239// abs(...)
3240// neg(...)
3241// sext(...)
3242// -reg
3243// -|...|
3244// -abs(...)
3245// name:...
3246// "name ::" is the VOPD separator, not an opcode modifier.
3247//
3248bool AMDGPUAsmParser::isModifier() {
3249
3250 AsmToken Tok = getToken();
3251 AsmToken NextToken[2];
3252 peekTokens(NextToken);
3253
3254 // "name:value" is an opcode modifier. The second colon of "::" is the
3255 // VOPD separator, so a symbol written immediately before "::" is a literal.
3256 bool IsOpcodeModifier = isOpcodeModifierWithVal(Tok, NextToken[0]) &&
3257 !NextToken[1].is(AsmToken::Colon);
3258
3259 return isOperandModifier(Tok, NextToken[0]) ||
3260 (Tok.is(AsmToken::Minus) &&
3261 isRegOrOperandModifier(NextToken[0], NextToken[1])) ||
3262 IsOpcodeModifier;
3263}
3264
3265// Check if the current token is an SP3 'neg' modifier.
3266// Currently this modifier is allowed in the following context:
3267//
3268// 1. Before a register, e.g. "-v0", "-v[...]" or "-[v0,v1]".
3269// 2. Before an 'abs' modifier: -abs(...)
3270// 3. Before an SP3 'abs' modifier: -|...|
3271//
3272// In all other cases "-" is handled as a part
3273// of an expression that follows the sign.
3274//
3275// Note: When "-" is followed by an integer literal,
3276// this is interpreted as integer negation rather
3277// than a floating-point NEG modifier applied to N.
3278// Beside being contr-intuitive, such use of floating-point
3279// NEG modifier would have resulted in different meaning
3280// of integer literals used with VOP1/2/C and VOP3,
3281// for example:
3282// v_exp_f32_e32 v5, -1 // VOP1: src0 = 0xFFFFFFFF
3283// v_exp_f32_e64 v5, -1 // VOP3: src0 = 0x80000001
3284// Negative fp literals with preceding "-" are
3285// handled likewise for uniformity
3286//
3287bool AMDGPUAsmParser::parseSP3NegModifier() {
3288
3289 AsmToken NextToken[2];
3290 peekTokens(NextToken);
3291
3292 if (isToken(AsmToken::Minus) &&
3293 (isRegister(NextToken[0], NextToken[1]) ||
3294 NextToken[0].is(AsmToken::Pipe) || isId(NextToken[0], "abs"))) {
3295 lex();
3296 return true;
3297 }
3298
3299 return false;
3300}
3301
3302ParseStatus
3303AMDGPUAsmParser::parseRegOrImmWithFPInputMods(OperandVector &Operands,
3304 bool AllowImm) {
3305 bool Neg, SP3Neg;
3306 bool Abs, SP3Abs;
3307 SMLoc Loc;
3308
3309 // Disable ambiguous constructs like '--1' etc. Should use neg(-1) instead.
3310 if (isToken(AsmToken::Minus) && peekToken().is(AsmToken::Minus))
3311 return Error(getLoc(), "invalid syntax, expected 'neg' modifier");
3312
3313 SP3Neg = parseSP3NegModifier();
3314
3315 Loc = getLoc();
3316 Neg = trySkipId("neg");
3317 if (Neg && SP3Neg)
3318 return Error(Loc, "expected register or immediate");
3319 if (Neg && !skipToken(AsmToken::LParen, "expected left paren after neg"))
3320 return ParseStatus::Failure;
3321
3322 Abs = trySkipId("abs");
3323 if (Abs && !skipToken(AsmToken::LParen, "expected left paren after abs"))
3324 return ParseStatus::Failure;
3325
3326 LitModifier Lit = LitModifier::None;
3327 if (trySkipId("lit")) {
3328 Lit = LitModifier::Lit;
3329 if (!skipToken(AsmToken::LParen, "expected left paren after lit"))
3330 return ParseStatus::Failure;
3331 } else if (trySkipId("lit64")) {
3332 Lit = LitModifier::Lit64;
3333 if (!skipToken(AsmToken::LParen, "expected left paren after lit64"))
3334 return ParseStatus::Failure;
3335 if (!has64BitLiterals())
3336 return Error(Loc, "lit64 is not supported on this GPU");
3337 }
3338
3339 Loc = getLoc();
3340 SP3Abs = trySkipToken(AsmToken::Pipe);
3341 if (Abs && SP3Abs)
3342 return Error(Loc, "expected register or immediate");
3343
3344 ParseStatus Res;
3345 if (AllowImm) {
3346 Res = parseRegOrImm(Operands, SP3Abs, Lit);
3347 } else {
3348 Res = parseReg(Operands);
3349 }
3350 if (!Res.isSuccess())
3351 return (SP3Neg || Neg || SP3Abs || Abs || Lit != LitModifier::None)
3353 : Res;
3354
3355 if (Lit != LitModifier::None && !Operands.back()->isImm())
3356 Error(Loc, "expected immediate with lit modifier");
3357
3358 if (SP3Abs && !skipToken(AsmToken::Pipe, "expected vertical bar"))
3359 return ParseStatus::Failure;
3360 if (Abs && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3361 return ParseStatus::Failure;
3362 if (Neg && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3363 return ParseStatus::Failure;
3364 if (Lit != LitModifier::None &&
3365 !skipToken(AsmToken::RParen, "expected closing parentheses"))
3366 return ParseStatus::Failure;
3367
3368 AMDGPUOperand::Modifiers Mods;
3369 Mods.Abs = Abs || SP3Abs;
3370 Mods.Neg = Neg || SP3Neg;
3371 Mods.Lit = Lit;
3372
3373 if (Mods.hasFPModifiers() || Lit != LitModifier::None) {
3374 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3375 if (Op.isExpr())
3376 return Error(Op.getStartLoc(), "expected an absolute expression");
3377 Op.setModifiers(Mods);
3378 }
3379 return ParseStatus::Success;
3380}
3381
3382ParseStatus
3383AMDGPUAsmParser::parseRegOrImmWithIntInputMods(OperandVector &Operands,
3384 bool AllowImm) {
3385 bool Sext = trySkipId("sext");
3386 if (Sext && !skipToken(AsmToken::LParen, "expected left paren after sext"))
3387 return ParseStatus::Failure;
3388
3389 ParseStatus Res;
3390 if (AllowImm) {
3391 Res = parseRegOrImm(Operands);
3392 } else {
3393 Res = parseReg(Operands);
3394 }
3395 if (!Res.isSuccess())
3396 return Sext ? ParseStatus::Failure : Res;
3397
3398 if (Sext && !skipToken(AsmToken::RParen, "expected closing parentheses"))
3399 return ParseStatus::Failure;
3400
3401 AMDGPUOperand::Modifiers Mods;
3402 Mods.Sext = Sext;
3403
3404 if (Mods.hasIntModifiers()) {
3405 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands.back());
3406 if (Op.isExpr())
3407 return Error(Op.getStartLoc(), "expected an absolute expression");
3408 Op.setModifiers(Mods);
3409 }
3410
3411 return ParseStatus::Success;
3412}
3413
3414ParseStatus AMDGPUAsmParser::parseRegWithFPInputMods(OperandVector &Operands) {
3415 return parseRegOrImmWithFPInputMods(Operands, false);
3416}
3417
3418ParseStatus AMDGPUAsmParser::parseRegWithIntInputMods(OperandVector &Operands) {
3419 return parseRegOrImmWithIntInputMods(Operands, false);
3420}
3421
3422ParseStatus AMDGPUAsmParser::parseRsrcReg(OperandVector &Operands) {
3423 // Without the marker, fall back to plain register parsing so the legacy
3424 // bare-register form (e.g. `s8`, `v8`) still assembles for indexed
3425 // buffer/image instructions.
3426 if (!trySkipId("rsrcidx"))
3427 return parseReg(Operands);
3428
3429 if (!skipToken(AsmToken::LParen, "expected left paren after rsrcidx"))
3430 return ParseStatus::Failure;
3431
3432 SMLoc RegLoc = getLoc();
3433 std::unique_ptr<AMDGPUOperand> Reg = parseRegister();
3434 if (!Reg)
3435 return ParseStatus::Failure;
3436
3437 // Enforce that the inner register is a valid index register. The matcher
3438 // predicate alone is not sufficient: if it fails, the matcher will fall back
3439 // to a non-indexed instruction variant whose resource operand happens to
3440 // accept the same register, silently dropping the `rsrcidx` intent.
3441 if (!Reg->isRsrcReg32())
3442 return Error(RegLoc, "rsrcidx operand must be a 32-bit SGPR or VGPR");
3443
3444 if (!skipToken(AsmToken::RParen, "expected closing parenthesis"))
3445 return ParseStatus::Failure;
3446
3447 Operands.push_back(std::move(Reg));
3448 return ParseStatus::Success;
3449}
3450
3451ParseStatus AMDGPUAsmParser::parseVReg32OrOff(OperandVector &Operands) {
3452 auto Loc = getLoc();
3453 if (trySkipId("off")) {
3454 Operands.push_back(
3455 AMDGPUOperand::CreateImm(this, 0, Loc, AMDGPUOperand::ImmTyOff, false));
3456 return ParseStatus::Success;
3457 }
3458
3459 if (!isRegister())
3460 return ParseStatus::NoMatch;
3461
3462 std::unique_ptr<AMDGPUOperand> Reg = parseRegister();
3463 if (Reg) {
3464 Operands.push_back(std::move(Reg));
3465 return ParseStatus::Success;
3466 }
3467
3468 return ParseStatus::Failure;
3469}
3470
3471unsigned AMDGPUAsmParser::checkTargetMatchPredicate(MCInst &Inst) {
3472 if ((getForcedEncodingSize() == 32 && SIInstrFlags::isVOP3(MII, Inst)) ||
3473 (getForcedEncodingSize() == 64 && !SIInstrFlags::isVOP3(MII, Inst)) ||
3474 (isForcedDPP() && !SIInstrFlags::isDPP(MII, Inst)) ||
3475 (isForcedSDWA() && !SIInstrFlags::isSDWA(MII, Inst)))
3476 return Match_InvalidOperand;
3477
3478 if (Inst.getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
3479 Inst.getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
3480 // v_mac_f32/16 allow only dst_sel == DWORD;
3481 auto OpNum =
3482 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::dst_sel);
3483 const auto &Op = Inst.getOperand(OpNum);
3484 if (!Op.isImm() || Op.getImm() != AMDGPU::SDWA::SdwaSel::DWORD) {
3485 return Match_InvalidOperand;
3486 }
3487 }
3488
3489 // Asm can first try to match VOPD or VOPD3. By failing early here with
3490 // Match_InvalidOperand, the parser will retry parsing as VOPD3 or VOPD.
3491 // Checking later during validateInstruction does not give a chance to retry
3492 // parsing as a different encoding.
3493 if (tryAnotherVOPDEncoding(Inst))
3494 return Match_InvalidOperand;
3495
3496 return Match_Success;
3497}
3498
3507
3508// What asm variants we should check
3509ArrayRef<unsigned> AMDGPUAsmParser::getMatchedVariants() const {
3510 if (isForcedDPP() && isForcedVOP3()) {
3511 static const unsigned Variants[] = {AMDGPUAsmVariants::VOP3_DPP};
3512 return ArrayRef(Variants);
3513 }
3514 if (getForcedEncodingSize() == 32) {
3515 static const unsigned Variants[] = {AMDGPUAsmVariants::DEFAULT};
3516 return ArrayRef(Variants);
3517 }
3518
3519 if (isForcedVOP3()) {
3520 static const unsigned Variants[] = {AMDGPUAsmVariants::VOP3};
3521 return ArrayRef(Variants);
3522 }
3523
3524 if (isForcedSDWA()) {
3525 static const unsigned Variants[] = {AMDGPUAsmVariants::SDWA,
3527 return ArrayRef(Variants);
3528 }
3529
3530 if (isForcedDPP()) {
3531 static const unsigned Variants[] = {AMDGPUAsmVariants::DPP};
3532 return ArrayRef(Variants);
3533 }
3534
3535 return getAllVariants();
3536}
3537
3538StringRef AMDGPUAsmParser::getMatchedVariantName() const {
3539 if (isForcedDPP() && isForcedVOP3())
3540 return "e64_dpp";
3541
3542 if (getForcedEncodingSize() == 32)
3543 return "e32";
3544
3545 if (isForcedVOP3())
3546 return "e64";
3547
3548 if (isForcedSDWA())
3549 return "sdwa";
3550
3551 if (isForcedDPP())
3552 return "dpp";
3553
3554 return "";
3555}
3556
3557MCRegister
3558AMDGPUAsmParser::findImplicitSGPRReadInVOP(const MCInst &Inst) const {
3559 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
3560 for (MCPhysReg Reg : Desc.implicit_uses()) {
3561 switch (Reg) {
3562 case AMDGPU::FLAT_SCR:
3563 case AMDGPU::VCC:
3564 case AMDGPU::VCC_LO:
3565 case AMDGPU::VCC_HI:
3566 case AMDGPU::M0:
3567 return Reg;
3568 default:
3569 break;
3570 }
3571 }
3572 return MCRegister();
3573}
3574
3575// NB: This code is correct only when used to check constant
3576// bus limitations because GFX7 support no f16 inline constants.
3577// Note that there are no cases when a GFX7 opcode violates
3578// constant bus limitations due to the use of an f16 constant.
3579bool AMDGPUAsmParser::isInlineConstant(const MCInst &Inst,
3580 unsigned OpIdx) const {
3581 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
3582
3583 if (!AMDGPU::isSISrcOperand(Desc, OpIdx) ||
3584 AMDGPU::isKImmOperand(Desc, OpIdx)) {
3585 return false;
3586 }
3587
3588 const MCOperand &MO = Inst.getOperand(OpIdx);
3589
3590 int64_t Val = MO.isImm() ? MO.getImm() : getLitValue(MO.getExpr());
3591 auto OpSize = AMDGPU::getOperandSize(Desc, OpIdx);
3592
3593 switch (OpSize) { // expected operand size
3594 case 8:
3595 return AMDGPU::isInlinableLiteral64(Val, hasInv2PiInlineImm());
3596 case 4:
3597 return AMDGPU::isInlinableLiteral32(Val, hasInv2PiInlineImm());
3598 case 2: {
3599 const unsigned OperandType = Desc.operands()[OpIdx].OperandType;
3602 return AMDGPU::isInlinableLiteralI16(Val, hasInv2PiInlineImm());
3603
3607
3611
3614
3618
3621 return AMDGPU::isInlinableLiteralFP16(Val, hasInv2PiInlineImm());
3622
3625 return AMDGPU::isInlinableLiteralBF16(Val, hasInv2PiInlineImm());
3626
3629 return false;
3630
3631 llvm_unreachable("invalid operand type");
3632 }
3633 default:
3634 llvm_unreachable("invalid operand size");
3635 }
3636}
3637
3638unsigned AMDGPUAsmParser::getConstantBusLimit(unsigned Opcode) const {
3639 if (!isGFX10Plus())
3640 return 1;
3641
3642 switch (Opcode) {
3643 // 64-bit shift instructions can use only one scalar value input
3644 case AMDGPU::V_LSHLREV_B64_e64:
3645 case AMDGPU::V_LSHLREV_B64_gfx10:
3646 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
3647 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
3648 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
3649 case AMDGPU::V_LSHRREV_B64_e64:
3650 case AMDGPU::V_LSHRREV_B64_gfx10:
3651 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
3652 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
3653 case AMDGPU::V_ASHRREV_I64_e64:
3654 case AMDGPU::V_ASHRREV_I64_gfx10:
3655 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
3656 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
3657 case AMDGPU::V_LSHL_B64_e64:
3658 case AMDGPU::V_LSHR_B64_e64:
3659 case AMDGPU::V_ASHR_I64_e64:
3660 return 1;
3661 default:
3662 return 2;
3663 }
3664}
3665
3666constexpr unsigned MAX_SRC_OPERANDS_NUM = 6;
3668
3669// Get regular operand indices in the same order as specified
3670// in the instruction (but append mandatory literals to the end).
3672 bool AddMandatoryLiterals = false) {
3673
3674 int16_t ImmIdx =
3675 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::imm) : -1;
3676
3677 if (isVOPD(Opcode)) {
3678 int16_t ImmXIdx =
3679 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::immX) : -1;
3680
3681 return {getNamedOperandIdx(Opcode, OpName::src0X),
3682 getNamedOperandIdx(Opcode, OpName::vsrc1X),
3683 getNamedOperandIdx(Opcode, OpName::vsrc2X),
3684 getNamedOperandIdx(Opcode, OpName::src0Y),
3685 getNamedOperandIdx(Opcode, OpName::vsrc1Y),
3686 getNamedOperandIdx(Opcode, OpName::vsrc2Y),
3687 ImmXIdx,
3688 ImmIdx};
3689 }
3690
3691 return {getNamedOperandIdx(Opcode, OpName::src0),
3692 getNamedOperandIdx(Opcode, OpName::src1),
3693 getNamedOperandIdx(Opcode, OpName::src2), ImmIdx};
3694}
3695
3696bool AMDGPUAsmParser::usesConstantBus(const MCInst &Inst, unsigned OpIdx) {
3697 const MCOperand &MO = Inst.getOperand(OpIdx);
3698 if (MO.isImm())
3699 return !isInlineConstant(Inst, OpIdx);
3700 if (MO.isReg()) {
3701 auto Reg = MO.getReg();
3702 if (!Reg)
3703 return false;
3704 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
3705 auto PReg = mc2PseudoReg(Reg);
3706 return isSGPR(PReg, TRI) && PReg != SGPR_NULL;
3707 }
3708 return true;
3709}
3710
3711// Based on the comment for `AMDGPUInstructionSelector::selectWritelane`:
3712// Writelane is special in that it can use SGPR and M0 (which would normally
3713// count as using the constant bus twice - but in this case it is allowed since
3714// the lane selector doesn't count as a use of the constant bus). However, it is
3715// still required to abide by the 1 SGPR rule.
3716static bool checkWriteLane(const MCInst &Inst) {
3717 const unsigned Opcode = Inst.getOpcode();
3718 if (Opcode != V_WRITELANE_B32_gfx6_gfx7 && Opcode != V_WRITELANE_B32_vi)
3719 return false;
3720 const MCOperand &LaneSelOp = Inst.getOperand(2);
3721 if (!LaneSelOp.isReg())
3722 return false;
3723 auto LaneSelReg = mc2PseudoReg(LaneSelOp.getReg());
3724 return LaneSelReg == M0 || LaneSelReg == M0_gfxpre11;
3725}
3726
3727bool AMDGPUAsmParser::validateConstantBusLimitations(
3728 const MCInst &Inst, const OperandVector &Operands) {
3729 const unsigned Opcode = Inst.getOpcode();
3730 const MCInstrDesc &Desc = MII.get(Opcode);
3731 MCRegister LastSGPR;
3732 unsigned ConstantBusUseCount = 0;
3733 unsigned NumLiterals = 0;
3734 unsigned LiteralSize;
3735
3738 !SIInstrFlags::isSDWA(Desc) && !isVOPD(Opcode))
3739 return true;
3740
3741 if (checkWriteLane(Inst))
3742 return true;
3743
3744 // Check special imm operands (used by madmk, etc)
3745 if (AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::imm)) {
3746 ++NumLiterals;
3747 LiteralSize = 4;
3748 }
3749
3750 SmallDenseSet<MCRegister> SGPRsUsed;
3751 MCRegister SGPRUsed = findImplicitSGPRReadInVOP(Inst);
3752 if (SGPRUsed) {
3753 SGPRsUsed.insert(SGPRUsed);
3754 ++ConstantBusUseCount;
3755 }
3756
3757 OperandIndices OpIndices = getSrcOperandIndices(Opcode);
3758
3759 unsigned ConstantBusLimit = getConstantBusLimit(Opcode);
3760
3761 for (int OpIdx : OpIndices) {
3762 if (OpIdx == -1)
3763 continue;
3764
3765 const MCOperand &MO = Inst.getOperand(OpIdx);
3766 if (usesConstantBus(Inst, OpIdx)) {
3767 if (MO.isReg()) {
3768 LastSGPR = mc2PseudoReg(MO.getReg());
3769 // Pairs of registers with a partial intersections like these
3770 // s0, s[0:1]
3771 // flat_scratch_lo, flat_scratch
3772 // flat_scratch_lo, flat_scratch_hi
3773 // are theoretically valid but they are disabled anyway.
3774 // Note that this code mimics SIInstrInfo::verifyInstruction
3775 if (SGPRsUsed.insert(LastSGPR).second) {
3776 ++ConstantBusUseCount;
3777 }
3778 } else { // Expression or a literal
3779
3780 if (Desc.operands()[OpIdx].OperandType == MCOI::OPERAND_IMMEDIATE)
3781 continue; // special operand like VINTERP attr_chan
3782
3783 // An instruction may use only one literal.
3784 // This has been validated on the previous step.
3785 // See validateVOPLiteral.
3786 // This literal may be used as more than one operand.
3787 // If all these operands are of the same size,
3788 // this literal counts as one scalar value.
3789 // Otherwise it counts as 2 scalar values.
3790 // See "GFX10 Shader Programming", section 3.6.2.3.
3791
3792 unsigned Size = AMDGPU::getOperandSize(Desc, OpIdx);
3793 if (Size < 4)
3794 Size = 4;
3795
3796 if (NumLiterals == 0) {
3797 NumLiterals = 1;
3798 LiteralSize = Size;
3799 } else if (LiteralSize != Size) {
3800 NumLiterals = 2;
3801 }
3802 }
3803 }
3804
3805 if (ConstantBusUseCount + NumLiterals > ConstantBusLimit) {
3806 Error(getOperandLoc(Operands, OpIdx),
3807 "invalid operand (violates constant bus restrictions)");
3808 return false;
3809 }
3810 }
3811 return true;
3812}
3813
3814std::optional<unsigned>
3815AMDGPUAsmParser::checkVOPDRegBankConstraints(const MCInst &Inst, bool AsVOPD3) {
3816
3817 const unsigned Opcode = Inst.getOpcode();
3818 if (!isVOPD(Opcode))
3819 return {};
3820
3821 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
3822
3823 auto getVRegIdx = [&](unsigned, unsigned OperandIdx) {
3824 const MCOperand &Opr = Inst.getOperand(OperandIdx);
3825 return (Opr.isReg() && !isSGPR(mc2PseudoReg(Opr.getReg()), TRI))
3826 ? Opr.getReg()
3827 : MCRegister();
3828 };
3829
3830 // On GFX1170+ if both OpX and OpY are V_MOV_B32 then OPY uses SRC2
3831 // source-cache.
3832 bool SkipSrc =
3833 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1170 ||
3834 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 ||
3835 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 ||
3836 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx13 ||
3837 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx1250 ||
3838 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx13;
3839 bool AllowSameVGPR = isGFX12Plus();
3840
3841 if (AsVOPD3) { // Literal constants are not allowed with VOPD3.
3842 for (auto OpName : {OpName::src0X, OpName::src0Y}) {
3843 int I = getNamedOperandIdx(Opcode, OpName);
3844 const MCOperand &Op = Inst.getOperand(I);
3845 if (!Op.isImm())
3846 continue;
3847 int64_t Imm = Op.getImm();
3848 if (!AMDGPU::isInlinableLiteral32(Imm, hasInv2PiInlineImm()) &&
3849 !AMDGPU::isInlinableLiteral64(Imm, hasInv2PiInlineImm()))
3850 return (unsigned)I;
3851 }
3852
3853 for (auto OpName : {OpName::vsrc1X, OpName::vsrc1Y, OpName::vsrc2X,
3854 OpName::vsrc2Y, OpName::imm}) {
3855 int I = getNamedOperandIdx(Opcode, OpName);
3856 if (I == -1)
3857 continue;
3858 const MCOperand &Op = Inst.getOperand(I);
3859 if (Op.isImm())
3860 return (unsigned)I;
3861 }
3862 }
3863
3864 const auto &InstInfo = getVOPDInstInfo(Opcode, &MII);
3865 auto InvalidCompOprIdx = InstInfo.getInvalidCompOperandIndex(
3866 getVRegIdx, *TRI, SkipSrc, AllowSameVGPR, AsVOPD3);
3867
3868 return InvalidCompOprIdx;
3869}
3870
3871bool AMDGPUAsmParser::validateVOPD(const MCInst &Inst,
3872 const OperandVector &Operands) {
3873
3874 unsigned Opcode = Inst.getOpcode();
3875 bool AsVOPD3 = SIInstrFlags::isVOPD3(MII, Inst);
3876
3877 if (AsVOPD3) {
3878 for (const std::unique_ptr<MCParsedAsmOperand> &Operand : Operands) {
3879 AMDGPUOperand &Op = (AMDGPUOperand &)*Operand;
3880 if ((Op.isRegKind() || Op.isImmTy(AMDGPUOperand::ImmTyNone)) &&
3881 (Op.getModifiers().getFPModifiersOperand() & SISrcMods::ABS))
3882 Error(Op.getStartLoc(), "ABS not allowed in VOPD3 instructions");
3883 }
3884 }
3885
3886 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, AsVOPD3);
3887 if (!InvalidCompOprIdx.has_value())
3888 return true;
3889
3890 auto CompOprIdx = *InvalidCompOprIdx;
3891 const auto &InstInfo = getVOPDInstInfo(Opcode, &MII);
3892 auto ParsedIdx =
3893 std::max(InstInfo[VOPD::X].getIndexInParsedOperands(CompOprIdx),
3894 InstInfo[VOPD::Y].getIndexInParsedOperands(CompOprIdx));
3895 assert(ParsedIdx > 0 && ParsedIdx < Operands.size());
3896
3897 auto Loc = ((AMDGPUOperand &)*Operands[ParsedIdx]).getStartLoc();
3898 if (CompOprIdx == VOPD::Component::DST) {
3899 if (AsVOPD3)
3900 Error(Loc, "dst registers must be distinct");
3901 else
3902 Error(Loc, "one dst register must be even and the other odd");
3903 } else {
3904 auto CompSrcIdx = CompOprIdx - VOPD::Component::DST_NUM;
3905 Error(Loc, Twine("src") + Twine(CompSrcIdx) +
3906 " operands must use different VGPR banks");
3907 }
3908
3909 return false;
3910}
3911
3912// \returns true if \p Inst does not satisfy VOPD constraints, but can be
3913// potentially used as VOPD3 with the same operands.
3914bool AMDGPUAsmParser::tryVOPD3(const MCInst &Inst) {
3915 // First check if it fits VOPD
3916 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, false);
3917 if (!InvalidCompOprIdx.has_value())
3918 return false;
3919
3920 // Then if it fits VOPD3
3921 InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, true);
3922 if (InvalidCompOprIdx.has_value()) {
3923 // If failed operand is dst it is better to show error about VOPD3
3924 // instruction as it has more capabilities and error message will be
3925 // more informative. If the dst is not legal for VOPD3, then it is not
3926 // legal for VOPD either.
3927 if (*InvalidCompOprIdx == VOPD::Component::DST)
3928 return true;
3929
3930 // Otherwise prefer VOPD as we may find ourselves in an awkward situation
3931 // with a conflict in tied implicit src2 of fmac and no asm operand to
3932 // to point to.
3933 return false;
3934 }
3935 return true;
3936}
3937
3938// \returns true is a VOPD3 instruction can be also represented as a shorter
3939// VOPD encoding.
3940bool AMDGPUAsmParser::tryVOPD(const MCInst &Inst) {
3941 const unsigned Opcode = Inst.getOpcode();
3942 const auto &II = getVOPDInstInfo(Opcode, &MII);
3943 unsigned EncodingFamily = AMDGPU::getVOPDEncodingFamily(getSTI());
3944 if (!getCanBeVOPD(II[VOPD::X].getOpcode(), EncodingFamily, false).X ||
3945 !getCanBeVOPD(II[VOPD::Y].getOpcode(), EncodingFamily, false).Y)
3946 return false;
3947
3948 // This is an awkward exception, VOPD3 variant of V_DUAL_CNDMASK_B32 has
3949 // explicit src2 even if it is vcc_lo. If it was parsed as VOPD3 it cannot
3950 // be parsed as VOPD which does not accept src2.
3951 if (II[VOPD::X].getOpcode() == AMDGPU::V_CNDMASK_B32_e32 ||
3952 II[VOPD::Y].getOpcode() == AMDGPU::V_CNDMASK_B32_e32)
3953 return false;
3954
3955 // If any modifiers are set this cannot be VOPD.
3956 for (auto OpName : {OpName::src0X_modifiers, OpName::src0Y_modifiers,
3957 OpName::vsrc1X_modifiers, OpName::vsrc1Y_modifiers,
3958 OpName::vsrc2X_modifiers, OpName::vsrc2Y_modifiers}) {
3959 int I = getNamedOperandIdx(Opcode, OpName);
3960 if (I == -1)
3961 continue;
3962 if (Inst.getOperand(I).getImm())
3963 return false;
3964 }
3965
3966 return !tryVOPD3(Inst);
3967}
3968
3969// VOPD3 has more relaxed register constraints than VOPD. We prefer shorter VOPD
3970// form but switch to VOPD3 otherwise.
3971bool AMDGPUAsmParser::tryAnotherVOPDEncoding(const MCInst &Inst) {
3972 if (!isGFX1250Plus() || !isVOPD(Inst.getOpcode()))
3973 return false;
3974
3975 if (SIInstrFlags::isVOPD3(MII, Inst))
3976 return tryVOPD(Inst);
3977 return tryVOPD3(Inst);
3978}
3979
3980bool AMDGPUAsmParser::validateIntClampSupported(const MCInst &Inst) {
3981
3982 const unsigned Opc = Inst.getOpcode();
3983
3984 if (SIInstrFlags::hasIntClamp(MII, Inst) && !hasIntClamp()) {
3985 int ClampIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::clamp);
3986 assert(ClampIdx != -1);
3987 return Inst.getOperand(ClampIdx).getImm() == 0;
3988 }
3989
3990 return true;
3991}
3992
3993bool AMDGPUAsmParser::validateMIMGDataSize(const MCInst &Inst, SMLoc IDLoc) {
3994
3995 const unsigned Opc = Inst.getOpcode();
3996 const MCInstrDesc &Desc = MII.get(Opc);
3997
3998 if ((SIInstrFlags::isImage(Desc)) == 0)
3999 return true;
4000
4001 int VDataIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdata);
4002 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4003 int TFEIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::tfe);
4004
4005 if (VDataIdx == -1 && isGFX10Plus()) // no return image_sample
4006 return true;
4007
4008 if ((DMaskIdx == -1 || TFEIdx == -1) &&
4009 hasBVHRayTracingInsts()) // intersect_ray
4010 return true;
4011
4012 unsigned VDataSize = getRegOperandSize(Desc, VDataIdx);
4013 unsigned TFESize = (TFEIdx != -1 && Inst.getOperand(TFEIdx).getImm()) ? 1 : 0;
4014 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4015 if (DMask == 0)
4016 DMask = 1;
4017
4018 bool IsPackedD16 = false;
4019 unsigned DataSize = SIInstrFlags::isGather4(Desc) ? 4 : llvm::popcount(DMask);
4020 if (hasPackedD16()) {
4021 int D16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::d16);
4022 IsPackedD16 = D16Idx >= 0;
4023 if (IsPackedD16 && Inst.getOperand(D16Idx).getImm())
4024 DataSize = (DataSize + 1) / 2;
4025 }
4026
4027 if ((VDataSize / 4) == DataSize + TFESize)
4028 return true;
4029
4030 StringRef Modifiers;
4031 if (isGFX90A())
4032 Modifiers = IsPackedD16 ? "dmask and d16" : "dmask";
4033 else
4034 Modifiers = IsPackedD16 ? "dmask, d16 and tfe" : "dmask and tfe";
4035
4036 Error(IDLoc, Twine("image data size does not match ") + Modifiers);
4037 return false;
4038}
4039
4040bool AMDGPUAsmParser::validateMIMGAddrSize(const MCInst &Inst, SMLoc IDLoc) {
4041 const unsigned Opc = Inst.getOpcode();
4042 const MCInstrDesc &Desc = MII.get(Opc);
4043
4045 return true;
4046
4047 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opc);
4048
4049 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4051 int VAddr0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr0);
4052 AMDGPU::OpName RSrcOpName =
4053 SIInstrFlags::isMIMG(Desc) ? AMDGPU::OpName::srsrc : AMDGPU::OpName::rsrc;
4054 int SrsrcIdx = AMDGPU::getNamedOperandIdx(Opc, RSrcOpName);
4055 int DimIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dim);
4056 int A16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::a16);
4057
4058 assert(VAddr0Idx != -1);
4059 assert(SrsrcIdx != -1);
4060 assert(SrsrcIdx > VAddr0Idx);
4061
4062 bool IsA16 = (A16Idx != -1 && Inst.getOperand(A16Idx).getImm());
4063 if (BaseOpcode->BVH) {
4064 if (IsA16 == BaseOpcode->A16)
4065 return true;
4066 Error(IDLoc, "image address size does not match a16");
4067 return false;
4068 }
4069
4070 unsigned Dim = Inst.getOperand(DimIdx).getImm();
4071 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByEncoding(Dim);
4072 bool IsNSA = SrsrcIdx - VAddr0Idx > 1;
4073 unsigned ActualAddrSize =
4074 IsNSA ? SrsrcIdx - VAddr0Idx : getRegOperandSize(Desc, VAddr0Idx) / 4;
4075
4076 unsigned ExpectedAddrSize =
4077 AMDGPU::getAddrSizeMIMGOp(BaseOpcode, DimInfo, IsA16, hasG16());
4078
4079 if (IsNSA) {
4080 if (hasPartialNSAEncoding() &&
4081 ExpectedAddrSize > getNSAMaxSize(SIInstrFlags::isVSAMPLE(Desc))) {
4082 int VAddrLastIdx = SrsrcIdx - 1;
4083 unsigned VAddrLastSize = getRegOperandSize(Desc, VAddrLastIdx) / 4;
4084
4085 ActualAddrSize = VAddrLastIdx - VAddr0Idx + VAddrLastSize;
4086 }
4087 } else {
4088 if (ExpectedAddrSize > 12)
4089 ExpectedAddrSize = 16;
4090
4091 // Allow oversized 8 VGPR vaddr when only 5/6/7 VGPRs are required.
4092 // This provides backward compatibility for assembly created
4093 // before 160b/192b/224b types were directly supported.
4094 if (ActualAddrSize == 8 && (ExpectedAddrSize >= 5 && ExpectedAddrSize <= 7))
4095 return true;
4096 }
4097
4098 if (ActualAddrSize == ExpectedAddrSize)
4099 return true;
4100
4101 Error(IDLoc, "image address size does not match dim and a16");
4102 return false;
4103}
4104
4105bool AMDGPUAsmParser::validateMIMGAtomicDMask(const MCInst &Inst) {
4106
4107 const unsigned Opc = Inst.getOpcode();
4108 const MCInstrDesc &Desc = MII.get(Opc);
4109
4110 if ((SIInstrFlags::isImage(Desc)) == 0)
4111 return true;
4112 if (!Desc.mayLoad() || !Desc.mayStore())
4113 return true; // Not atomic
4114
4115 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4116 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4117
4118 // This is an incomplete check because image_atomic_cmpswap
4119 // may only use 0x3 and 0xf while other atomic operations
4120 // may use 0x1 and 0x3. However these limitations are
4121 // verified when we check that dmask matches dst size.
4122 return DMask == 0x1 || DMask == 0x3 || DMask == 0xf;
4123}
4124
4125bool AMDGPUAsmParser::validateMIMGGatherDMask(const MCInst &Inst) {
4126
4127 const unsigned Opc = Inst.getOpcode();
4128
4129 if (!SIInstrFlags::isGather4(MII, Inst))
4130 return true;
4131
4132 int DMaskIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dmask);
4133 unsigned DMask = Inst.getOperand(DMaskIdx).getImm() & 0xf;
4134
4135 // GATHER4 instructions use dmask in a different fashion compared to
4136 // other MIMG instructions. The only useful DMASK values are
4137 // 1=red, 2=green, 4=blue, 8=alpha. (e.g. 1 returns
4138 // (red,red,red,red) etc.) The ISA document doesn't mention
4139 // this.
4140 return DMask == 0x1 || DMask == 0x2 || DMask == 0x4 || DMask == 0x8;
4141}
4142
4143bool AMDGPUAsmParser::validateMIMGDim(const MCInst &Inst,
4144 const OperandVector &Operands) {
4145 if (!isGFX10Plus())
4146 return true;
4147
4148 const unsigned Opc = Inst.getOpcode();
4149
4150 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4151 return true;
4152
4153 // image_bvh_intersect_ray instructions do not have dim
4155 return true;
4156
4157 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
4158 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4159 if (Op.isDim())
4160 return true;
4161 }
4162 return false;
4163}
4164
4165bool AMDGPUAsmParser::validateMIMGMSAA(const MCInst &Inst) {
4166 const unsigned Opc = Inst.getOpcode();
4167
4168 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4169 return true;
4170
4171 const AMDGPU::MIMGInfo *Info = AMDGPU::getMIMGInfo(Opc);
4172 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4174
4175 if (!BaseOpcode->MSAA)
4176 return true;
4177
4178 int DimIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dim);
4179 assert(DimIdx != -1);
4180
4181 unsigned Dim = Inst.getOperand(DimIdx).getImm();
4182 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByEncoding(Dim);
4183
4184 return DimInfo->MSAA;
4185}
4186
4187static bool IsMovrelsSDWAOpcode(const unsigned Opcode) {
4188 switch (Opcode) {
4189 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
4190 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
4191 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
4192 return true;
4193 default:
4194 return false;
4195 }
4196}
4197
4198// movrels* opcodes should only allow VGPRS as src0.
4199// This is specified in .td description for vop1/vop3,
4200// but sdwa is handled differently. See isSDWAOperand.
4201bool AMDGPUAsmParser::validateMovrels(const MCInst &Inst,
4202 const OperandVector &Operands) {
4203
4204 const unsigned Opc = Inst.getOpcode();
4205
4206 if (!SIInstrFlags::isSDWA(MII, Inst) || !IsMovrelsSDWAOpcode(Opc))
4207 return true;
4208
4209 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4210 assert(Src0Idx != -1);
4211
4212 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4213 if (Src0.isReg()) {
4214 auto Reg = mc2PseudoReg(Src0.getReg());
4215 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4216 if (!isSGPR(Reg, TRI))
4217 return true;
4218 }
4219
4220 Error(getOperandLoc(Operands, Src0Idx), "source operand must be a VGPR");
4221 return false;
4222}
4223
4224bool AMDGPUAsmParser::validateMAIAccWrite(const MCInst &Inst,
4225 const OperandVector &Operands) {
4226
4227 const unsigned Opc = Inst.getOpcode();
4228
4229 if (Opc != AMDGPU::V_ACCVGPR_WRITE_B32_vi)
4230 return true;
4231
4232 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4233 assert(Src0Idx != -1);
4234
4235 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4236 if (!Src0.isReg())
4237 return true;
4238
4239 auto Reg = mc2PseudoReg(Src0.getReg());
4240 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4241 if (!isGFX90A() && isSGPR(Reg, TRI)) {
4242 Error(getOperandLoc(Operands, Src0Idx),
4243 "source operand must be either a VGPR or an inline constant");
4244 return false;
4245 }
4246
4247 return true;
4248}
4249
4250bool AMDGPUAsmParser::validateMAISrc2(const MCInst &Inst,
4251 const OperandVector &Operands) {
4252 unsigned Opcode = Inst.getOpcode();
4253
4254 if (!SIInstrFlags::isMAI(MII, Inst) ||
4255 !getFeatureBits()[FeatureMFMAInlineLiteralBug])
4256 return true;
4257
4258 const int Src2Idx = getNamedOperandIdx(Opcode, OpName::src2);
4259 if (Src2Idx == -1)
4260 return true;
4261
4262 if (Inst.getOperand(Src2Idx).isImm() && isInlineConstant(Inst, Src2Idx)) {
4263 Error(getOperandLoc(Operands, Src2Idx),
4264 "inline constants are not allowed for this operand");
4265 return false;
4266 }
4267
4268 return true;
4269}
4270
4271bool AMDGPUAsmParser::validateMFMA(const MCInst &Inst,
4272 const OperandVector &Operands) {
4273 const unsigned Opc = Inst.getOpcode();
4274 const MCInstrDesc &Desc = MII.get(Opc);
4275
4277 return true;
4278
4279 int BlgpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
4280 if (BlgpIdx != -1) {
4281 if (const MFMA_F8F6F4_Info *Info = AMDGPU::isMFMA_F8F6F4(Opc)) {
4282 int CbszIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::cbsz);
4283
4284 unsigned CBSZ = Inst.getOperand(CbszIdx).getImm();
4285 unsigned BLGP = Inst.getOperand(BlgpIdx).getImm();
4286
4287 // Validate the correct register size was used for the floating point
4288 // format operands
4289
4290 bool Success = true;
4291 if (Info->NumRegsSrcA != mfmaScaleF8F6F4FormatToNumRegs(CBSZ)) {
4292 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4293 Error(getOperandLoc(Operands, Src0Idx),
4294 "wrong register tuple size for cbsz value " + Twine(CBSZ));
4295 Success = false;
4296 }
4297
4298 if (Info->NumRegsSrcB != mfmaScaleF8F6F4FormatToNumRegs(BLGP)) {
4299 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
4300 Error(getOperandLoc(Operands, Src1Idx),
4301 "wrong register tuple size for blgp value " + Twine(BLGP));
4302 Success = false;
4303 }
4304
4305 return Success;
4306 }
4307 }
4308
4309 const int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
4310 if (Src2Idx == -1)
4311 return true;
4312
4313 const MCOperand &Src2 = Inst.getOperand(Src2Idx);
4314 if (!Src2.isReg())
4315 return true;
4316
4317 MCRegister Src2Reg = Src2.getReg();
4318 MCRegister DstReg = Inst.getOperand(0).getReg();
4319 if (Src2Reg == DstReg)
4320 return true;
4321
4322 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4323 if (TRI->getRegClass(MII.getOpRegClassID(Desc.operands()[0], HwMode))
4324 .getSizeInBits() <= 128)
4325 return true;
4326
4327 if (TRI->regsOverlap(Src2Reg, DstReg)) {
4328 Error(getOperandLoc(Operands, Src2Idx),
4329 "source 2 operand must not partially overlap with dst");
4330 return false;
4331 }
4332
4333 return true;
4334}
4335
4336bool AMDGPUAsmParser::validateDivScale(const MCInst &Inst) {
4337 switch (Inst.getOpcode()) {
4338 default:
4339 return true;
4340 case V_DIV_SCALE_F32_gfx6_gfx7:
4341 case V_DIV_SCALE_F32_vi:
4342 case V_DIV_SCALE_F32_gfx10:
4343 case V_DIV_SCALE_F64_gfx6_gfx7:
4344 case V_DIV_SCALE_F64_vi:
4345 case V_DIV_SCALE_F64_gfx10:
4346 break;
4347 }
4348
4349 // TODO: Check that src0 = src1 or src2.
4350
4351 for (auto Name :
4352 {AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src2_modifiers,
4353 AMDGPU::OpName::src2_modifiers}) {
4354 if (Inst.getOperand(AMDGPU::getNamedOperandIdx(Inst.getOpcode(), Name))
4355 .getImm() &
4357 return false;
4358 }
4359 }
4360
4361 return true;
4362}
4363
4364bool AMDGPUAsmParser::validateMIMGD16(const MCInst &Inst) {
4365
4366 const unsigned Opc = Inst.getOpcode();
4367
4368 if ((SIInstrFlags::isImage(MII, Inst)) == 0)
4369 return true;
4370
4371 int D16Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::d16);
4372 if (D16Idx >= 0 && Inst.getOperand(D16Idx).getImm()) {
4373 if (isCI() || isSI())
4374 return false;
4375 }
4376
4377 return true;
4378}
4379
4380bool AMDGPUAsmParser::validateTensorR128(const MCInst &Inst) {
4381 const unsigned Opc = Inst.getOpcode();
4382
4383 if (!SIInstrFlags::usesTENSOR_CNT(MII, Inst))
4384 return true;
4385
4386 int R128Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::r128);
4387
4388 return R128Idx < 0 || !Inst.getOperand(R128Idx).getImm();
4389}
4390
4391static bool IsRevOpcode(const unsigned Opcode) {
4392 switch (Opcode) {
4393 case AMDGPU::V_SUBREV_F32_e32:
4394 case AMDGPU::V_SUBREV_F32_e64:
4395 case AMDGPU::V_SUBREV_F32_e32_gfx10:
4396 case AMDGPU::V_SUBREV_F32_e32_gfx6_gfx7:
4397 case AMDGPU::V_SUBREV_F32_e32_vi:
4398 case AMDGPU::V_SUBREV_F32_e64_gfx10:
4399 case AMDGPU::V_SUBREV_F32_e64_gfx6_gfx7:
4400 case AMDGPU::V_SUBREV_F32_e64_vi:
4401
4402 case AMDGPU::V_SUBREV_CO_U32_e32:
4403 case AMDGPU::V_SUBREV_CO_U32_e64:
4404 case AMDGPU::V_SUBREV_I32_e32_gfx6_gfx7:
4405 case AMDGPU::V_SUBREV_I32_e64_gfx6_gfx7:
4406
4407 case AMDGPU::V_SUBBREV_U32_e32:
4408 case AMDGPU::V_SUBBREV_U32_e64:
4409 case AMDGPU::V_SUBBREV_U32_e32_gfx6_gfx7:
4410 case AMDGPU::V_SUBBREV_U32_e32_vi:
4411 case AMDGPU::V_SUBBREV_U32_e64_gfx6_gfx7:
4412 case AMDGPU::V_SUBBREV_U32_e64_vi:
4413
4414 case AMDGPU::V_SUBREV_U32_e32:
4415 case AMDGPU::V_SUBREV_U32_e64:
4416 case AMDGPU::V_SUBREV_U32_e32_gfx9:
4417 case AMDGPU::V_SUBREV_U32_e32_vi:
4418 case AMDGPU::V_SUBREV_U32_e64_gfx9:
4419 case AMDGPU::V_SUBREV_U32_e64_vi:
4420
4421 case AMDGPU::V_SUBREV_F16_e32:
4422 case AMDGPU::V_SUBREV_F16_e64:
4423 case AMDGPU::V_SUBREV_F16_e32_gfx10:
4424 case AMDGPU::V_SUBREV_F16_e32_vi:
4425 case AMDGPU::V_SUBREV_F16_e64_gfx10:
4426 case AMDGPU::V_SUBREV_F16_e64_vi:
4427
4428 case AMDGPU::V_SUBREV_U16_e32:
4429 case AMDGPU::V_SUBREV_U16_e64:
4430 case AMDGPU::V_SUBREV_U16_e32_vi:
4431 case AMDGPU::V_SUBREV_U16_e64_vi:
4432
4433 case AMDGPU::V_SUBREV_CO_U32_e32_gfx9:
4434 case AMDGPU::V_SUBREV_CO_U32_e64_gfx10:
4435 case AMDGPU::V_SUBREV_CO_U32_e64_gfx9:
4436
4437 case AMDGPU::V_SUBBREV_CO_U32_e32_gfx9:
4438 case AMDGPU::V_SUBBREV_CO_U32_e64_gfx9:
4439
4440 case AMDGPU::V_SUBREV_NC_U32_e32_gfx10:
4441 case AMDGPU::V_SUBREV_NC_U32_e64_gfx10:
4442
4443 case AMDGPU::V_SUBREV_CO_CI_U32_e32_gfx10:
4444 case AMDGPU::V_SUBREV_CO_CI_U32_e64_gfx10:
4445
4446 case AMDGPU::V_LSHRREV_B32_e32:
4447 case AMDGPU::V_LSHRREV_B32_e64:
4448 case AMDGPU::V_LSHRREV_B32_e32_gfx6_gfx7:
4449 case AMDGPU::V_LSHRREV_B32_e64_gfx6_gfx7:
4450 case AMDGPU::V_LSHRREV_B32_e32_vi:
4451 case AMDGPU::V_LSHRREV_B32_e64_vi:
4452 case AMDGPU::V_LSHRREV_B32_e32_gfx10:
4453 case AMDGPU::V_LSHRREV_B32_e64_gfx10:
4454
4455 case AMDGPU::V_ASHRREV_I32_e32:
4456 case AMDGPU::V_ASHRREV_I32_e64:
4457 case AMDGPU::V_ASHRREV_I32_e32_gfx10:
4458 case AMDGPU::V_ASHRREV_I32_e32_gfx6_gfx7:
4459 case AMDGPU::V_ASHRREV_I32_e32_vi:
4460 case AMDGPU::V_ASHRREV_I32_e64_gfx10:
4461 case AMDGPU::V_ASHRREV_I32_e64_gfx6_gfx7:
4462 case AMDGPU::V_ASHRREV_I32_e64_vi:
4463
4464 case AMDGPU::V_LSHLREV_B32_e32:
4465 case AMDGPU::V_LSHLREV_B32_e64:
4466 case AMDGPU::V_LSHLREV_B32_e32_gfx10:
4467 case AMDGPU::V_LSHLREV_B32_e32_gfx6_gfx7:
4468 case AMDGPU::V_LSHLREV_B32_e32_vi:
4469 case AMDGPU::V_LSHLREV_B32_e64_gfx10:
4470 case AMDGPU::V_LSHLREV_B32_e64_gfx6_gfx7:
4471 case AMDGPU::V_LSHLREV_B32_e64_vi:
4472
4473 case AMDGPU::V_LSHLREV_B16_e32:
4474 case AMDGPU::V_LSHLREV_B16_e64:
4475 case AMDGPU::V_LSHLREV_B16_e32_vi:
4476 case AMDGPU::V_LSHLREV_B16_e64_vi:
4477 case AMDGPU::V_LSHLREV_B16_gfx10:
4478
4479 case AMDGPU::V_LSHRREV_B16_e32:
4480 case AMDGPU::V_LSHRREV_B16_e64:
4481 case AMDGPU::V_LSHRREV_B16_e32_vi:
4482 case AMDGPU::V_LSHRREV_B16_e64_vi:
4483 case AMDGPU::V_LSHRREV_B16_gfx10:
4484
4485 case AMDGPU::V_ASHRREV_I16_e32:
4486 case AMDGPU::V_ASHRREV_I16_e64:
4487 case AMDGPU::V_ASHRREV_I16_e32_vi:
4488 case AMDGPU::V_ASHRREV_I16_e64_vi:
4489 case AMDGPU::V_ASHRREV_I16_gfx10:
4490
4491 case AMDGPU::V_LSHLREV_B64_e64:
4492 case AMDGPU::V_LSHLREV_B64_gfx10:
4493 case AMDGPU::V_LSHLREV_B64_vi:
4494
4495 case AMDGPU::V_LSHRREV_B64_e64:
4496 case AMDGPU::V_LSHRREV_B64_gfx10:
4497 case AMDGPU::V_LSHRREV_B64_vi:
4498
4499 case AMDGPU::V_ASHRREV_I64_e64:
4500 case AMDGPU::V_ASHRREV_I64_gfx10:
4501 case AMDGPU::V_ASHRREV_I64_vi:
4502
4503 case AMDGPU::V_PK_LSHLREV_B16:
4504 case AMDGPU::V_PK_LSHLREV_B16_gfx10:
4505 case AMDGPU::V_PK_LSHLREV_B16_vi:
4506
4507 case AMDGPU::V_PK_LSHRREV_B16:
4508 case AMDGPU::V_PK_LSHRREV_B16_gfx10:
4509 case AMDGPU::V_PK_LSHRREV_B16_vi:
4510 case AMDGPU::V_PK_ASHRREV_I16:
4511 case AMDGPU::V_PK_ASHRREV_I16_gfx10:
4512 case AMDGPU::V_PK_ASHRREV_I16_vi:
4513 return true;
4514 default:
4515 return false;
4516 }
4517}
4518
4519bool AMDGPUAsmParser::validateLdsDirect(const MCInst &Inst,
4520 const OperandVector &Operands) {
4521 const unsigned Opcode = Inst.getOpcode();
4522
4523 // lds_direct register is defined so that it can be used
4524 // with 9-bit operands only. Ignore encodings which do not accept these.
4525 if (!SIInstrFlags::isVOP1(MII, Inst) && !SIInstrFlags::isVOP2(MII, Inst) &&
4526 !SIInstrFlags::isVOP3Like(MII, Inst) &&
4527 !SIInstrFlags::isVOPC(MII, Inst) && !SIInstrFlags::isSDWA(MII, Inst))
4528 return true;
4529
4530 for (auto SrcName : {OpName::src0, OpName::src1, OpName::src2}) {
4531 auto SrcIdx = getNamedOperandIdx(Opcode, SrcName);
4532 if (SrcIdx == -1)
4533 break;
4534 const auto &Src = Inst.getOperand(SrcIdx);
4535 if (Src.isReg() && Src.getReg() == LDS_DIRECT) {
4536
4537 if (isGFX90A() || isGFX11Plus()) {
4538 Error(getOperandLoc(Operands, SrcIdx),
4539 "lds_direct is not supported on this GPU");
4540 return false;
4541 }
4542
4543 if (IsRevOpcode(Opcode) || SIInstrFlags::isSDWA(MII, Inst)) {
4544 Error(getOperandLoc(Operands, SrcIdx),
4545 "lds_direct cannot be used with this instruction");
4546 return false;
4547 }
4548
4549 if (SrcName != OpName::src0) {
4550 Error(getOperandLoc(Operands, SrcIdx),
4551 "lds_direct may be used as src0 only");
4552 return false;
4553 }
4554 }
4555 }
4556
4557 return true;
4558}
4559
4560SMLoc AMDGPUAsmParser::getFlatOffsetLoc(const OperandVector &Operands) const {
4561 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
4562 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4563 if (Op.isFlatOffset())
4564 return Op.getStartLoc();
4565 }
4566 return getLoc();
4567}
4568
4569bool AMDGPUAsmParser::validateOffset(const MCInst &Inst,
4570 const OperandVector &Operands) {
4571 auto Opcode = Inst.getOpcode();
4572 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4573 if (OpNum == -1)
4574 return true;
4575
4576 if (SIInstrFlags::isFLAT(MII, Inst))
4577 return validateFlatOffset(Inst, Operands);
4578
4579 if (SIInstrFlags::isSMRD(MII, Inst))
4580 return validateSMEMOffset(Inst, Operands);
4581
4582 const auto &Op = Inst.getOperand(OpNum);
4583 // GFX12+ buffer ops: InstOffset is signed 24, but must not be a negative.
4584 if (isGFX12Plus() && SIInstrFlags::isBuffer(MII, Inst)) {
4585 const unsigned OffsetSize = 24;
4586 if (!isUIntN(OffsetSize - 1, Op.getImm())) {
4587 Error(getFlatOffsetLoc(Operands),
4588 Twine("expected a ") + Twine(OffsetSize - 1) +
4589 "-bit unsigned offset for buffer ops");
4590 return false;
4591 }
4592 } else {
4593 const unsigned OffsetSize = 16;
4594 if (!isUIntN(OffsetSize, Op.getImm())) {
4595 Error(getFlatOffsetLoc(Operands),
4596 Twine("expected a ") + Twine(OffsetSize) + "-bit unsigned offset");
4597 return false;
4598 }
4599 }
4600 return true;
4601}
4602
4603bool AMDGPUAsmParser::validateFlatOffset(const MCInst &Inst,
4604 const OperandVector &Operands) {
4605 if (!SIInstrFlags::isFLAT(MII, Inst))
4606 return true;
4607
4608 auto Opcode = Inst.getOpcode();
4609 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4610 assert(OpNum != -1);
4611
4612 const auto &Op = Inst.getOperand(OpNum);
4613 if (!hasFlatOffsets() && Op.getImm() != 0) {
4614 Error(getFlatOffsetLoc(Operands),
4615 "flat offset modifier is not supported on this GPU");
4616 return false;
4617 }
4618
4619 // For pre-GFX12 FLAT instructions the offset must be positive;
4620 // MSB is ignored and forced to zero.
4621 unsigned OffsetSize = AMDGPU::getNumFlatOffsetBits(getSTI());
4622 bool AllowNegative =
4624 if (!isIntN(OffsetSize, Op.getImm()) || (!AllowNegative && Op.getImm() < 0)) {
4625 Error(getFlatOffsetLoc(Operands),
4626 Twine("expected a ") +
4627 (AllowNegative ? Twine(OffsetSize) + "-bit signed offset"
4628 : Twine(OffsetSize - 1) + "-bit unsigned offset"));
4629 return false;
4630 }
4631
4632 return true;
4633}
4634
4635SMLoc AMDGPUAsmParser::getSMEMOffsetLoc(const OperandVector &Operands) const {
4636 // Start with second operand because SMEM Offset cannot be dst or src0.
4637 for (unsigned i = 2, e = Operands.size(); i != e; ++i) {
4638 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
4639 if (Op.isSMEMOffset() || Op.isSMEMOffsetMod())
4640 return Op.getStartLoc();
4641 }
4642 return getLoc();
4643}
4644
4645bool AMDGPUAsmParser::validateSMEMOffset(const MCInst &Inst,
4646 const OperandVector &Operands) {
4647 if (isCI() || isSI())
4648 return true;
4649
4650 if (!SIInstrFlags::isSMRD(MII, Inst))
4651 return true;
4652
4653 auto Opcode = Inst.getOpcode();
4654 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4655 if (OpNum == -1)
4656 return true;
4657
4658 const auto &Op = Inst.getOperand(OpNum);
4659 if (!Op.isImm())
4660 return true;
4661
4662 uint64_t Offset = Op.getImm();
4663 bool IsBuffer = AMDGPU::getSMEMIsBuffer(Opcode);
4666 return true;
4667
4668 Error(getSMEMOffsetLoc(Operands),
4669 isGFX12Plus() && IsBuffer
4670 ? "expected a 23-bit unsigned offset for buffer ops"
4671 : isGFX12Plus() ? "expected a 24-bit signed offset"
4672 : (isVI() || IsBuffer) ? "expected a 20-bit unsigned offset"
4673 : "expected a 21-bit signed offset");
4674
4675 return false;
4676}
4677
4678// On subtargets with FeatureBF16InlineConstFromUpperFP32 the hardware generates
4679// a bf16 inline constant in the high half of the corresponding fp32 inline
4680// constant. A VOP1 bf16 opcode (v_cvt_f32_bf16 and the bf16 transcendentals)
4681// reads the low half of its source, so it must use the VOP3 encoding with
4682// op_sel[0] set in order to see the constant at all.
4683bool AMDGPUAsmParser::validateBF16InlineConst(const MCInst &Inst,
4684 const OperandVector &Operands) {
4685 if (!getFeatureBits()[AMDGPU::FeatureBF16InlineConstFromUpperFP32])
4686 return true;
4687
4688 const unsigned Opc = Inst.getOpcode();
4689 const MCInstrDesc &Desc = MII.get(Opc);
4690 const bool IsVOP3 =
4692 if (!SIInstrFlags::isVOP1(Desc) && !IsVOP3)
4693 return true;
4694
4695 // Only the single-source VOP1 bf16 opcodes are affected. Multi-source and
4696 // packed bf16 instructions such as v_fma_mix*_bf16 are not.
4697 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::src1))
4698 return true;
4699
4700 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4701 if (Src0Idx == -1)
4702 return true;
4703
4704 const MCOperandInfo &Src0Info = Desc.operands()[Src0Idx];
4705 if (!AMDGPU::isBF16SrcOperand(Src0Info))
4706 return true;
4707
4708 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4709 if (!Src0.isImm() ||
4710 !AMDGPU::isInlinableLiteralBF16(static_cast<int16_t>(Src0.getImm()),
4711 hasInv2PiInlineImm()))
4712 return true;
4713
4714 if (IsVOP3) {
4715 const int ModsIdx =
4716 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
4717 if (ModsIdx != -1 &&
4718 (Inst.getOperand(ModsIdx).getImm() & SISrcMods::OP_SEL_0))
4719 return true;
4720 }
4721
4722 Error(getOperandLoc(Operands, Src0Idx),
4723 "bf16 inline constant is read from the high half of the fp32 inline "
4724 "constant on this GPU; use the e64 encoding with op_sel:[1,0]");
4725 return false;
4726}
4727
4728bool AMDGPUAsmParser::validateSOPLiteral(const MCInst &Inst,
4729 const OperandVector &Operands) {
4730 unsigned Opcode = Inst.getOpcode();
4731 const MCInstrDesc &Desc = MII.get(Opcode);
4733 return true;
4734
4735 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
4736 const int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
4737
4738 const int OpIndices[] = {Src0Idx, Src1Idx};
4739
4740 unsigned NumExprs = 0;
4741 unsigned NumLiterals = 0;
4742 int64_t LiteralValue;
4743
4744 for (int OpIdx : OpIndices) {
4745 if (OpIdx == -1)
4746 break;
4747
4748 const MCOperand &MO = Inst.getOperand(OpIdx);
4749 // Exclude special imm operands (like that used by s_set_gpr_idx_on)
4750 if (AMDGPU::isSISrcOperand(Desc, OpIdx)) {
4751 bool IsLit = false;
4752 std::optional<int64_t> Imm;
4753 if (MO.isImm()) {
4754 Imm = MO.getImm();
4755 } else if (MO.isExpr()) {
4756 if (isLitExpr(MO.getExpr())) {
4757 IsLit = true;
4758 Imm = getLitValue(MO.getExpr());
4759 }
4760 } else {
4761 continue;
4762 }
4763
4764 if (!Imm.has_value()) {
4765 ++NumExprs;
4766 } else if (!isInlineConstant(Inst, OpIdx)) {
4767 auto OpType = static_cast<AMDGPU::OperandType>(
4768 Desc.operands()[OpIdx].OperandType);
4769 int64_t Value = encode32BitLiteral(*Imm, OpType, IsLit);
4770 if (NumLiterals == 0 || LiteralValue != Value) {
4772 ++NumLiterals;
4773 }
4774 }
4775 }
4776 }
4777
4778 if (NumLiterals + NumExprs <= 1)
4779 return true;
4780
4781 Error(getOperandLoc(Operands, Src1Idx),
4782 "only one unique literal operand is allowed");
4783 return false;
4784}
4785
4786bool AMDGPUAsmParser::validateOpSel(const MCInst &Inst) {
4787 const unsigned Opc = Inst.getOpcode();
4788 if (isPermlane16(Opc)) {
4789 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
4790 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
4791
4792 if (OpSel & ~3)
4793 return false;
4794 }
4795
4796 if (isGFX940() && SIInstrFlags::isDOT(MII, Inst)) {
4797 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
4798 if (OpSelIdx != -1) {
4799 if (Inst.getOperand(OpSelIdx).getImm() != 0)
4800 return false;
4801 }
4802 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
4803 if (OpSelHiIdx != -1) {
4804 if (Inst.getOperand(OpSelHiIdx).getImm() != -1)
4805 return false;
4806 }
4807 }
4808
4809 // op_sel[0:1] must be 0 for v_dot2_bf16_bf16 and v_dot2_f16_f16 (VOP3 Dot).
4810 if (isGFX11Plus() && SIInstrFlags::isDOT(MII, Inst) &&
4811 SIInstrFlags::isVOP3(MII, Inst) && !SIInstrFlags::isVOP3P(MII, Inst)) {
4812 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
4813 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
4814 if (OpSel & 3)
4815 return false;
4816 }
4817
4818 // Packed math FP32 instructions typically accept SGPRs or VGPRs as source
4819 // operands. On gfx12+, if a source operand uses SGPRs, the HW can only read
4820 // the first SGPR and use it for both the low and high operations.
4822 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
4823 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
4824 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
4825 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
4826
4827 const MCOperand &Src0 = Inst.getOperand(Src0Idx);
4828 const MCOperand &Src1 = Inst.getOperand(Src1Idx);
4829 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
4830 unsigned OpSelHi = Inst.getOperand(OpSelHiIdx).getImm();
4831
4832 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4833
4834 auto VerifyOneSGPR = [OpSel, OpSelHi](unsigned Index) -> bool {
4835 unsigned Mask = 1U << Index;
4836 return ((OpSel & Mask) == 0) && ((OpSelHi & Mask) == 0);
4837 };
4838
4839 if (Src0.isReg() && isSGPR(Src0.getReg(), TRI) &&
4840 !VerifyOneSGPR(/*Index=*/0))
4841 return false;
4842 if (Src1.isReg() && isSGPR(Src1.getReg(), TRI) &&
4843 !VerifyOneSGPR(/*Index=*/1))
4844 return false;
4845
4846 int Src2Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2);
4847 if (Src2Idx != -1) {
4848 const MCOperand &Src2 = Inst.getOperand(Src2Idx);
4849 if (Src2.isReg() && isSGPR(Src2.getReg(), TRI) &&
4850 !VerifyOneSGPR(/*Index=*/2))
4851 return false;
4852 }
4853 }
4854
4855 return true;
4856}
4857
4858bool AMDGPUAsmParser::validateTrue16OpSel(const MCInst &Inst) {
4859 if (!hasTrue16Insts())
4860 return true;
4861 const MCRegisterInfo *MRI = getMRI();
4862 const unsigned Opc = Inst.getOpcode();
4863 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
4864 if (OpSelIdx == -1)
4865 return true;
4866 unsigned OpSelOpValue = Inst.getOperand(OpSelIdx).getImm();
4867 // If the value is 0 we could have a default OpSel Operand, so conservatively
4868 // allow it.
4869 if (OpSelOpValue == 0)
4870 return true;
4871 unsigned OpCount = 0;
4872 for (AMDGPU::OpName OpName : {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
4873 AMDGPU::OpName::src2, AMDGPU::OpName::vdst}) {
4874 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.getOpcode(), OpName);
4875 if (OpIdx == -1)
4876 continue;
4877 const MCOperand &Op = Inst.getOperand(OpIdx);
4878 if (Op.isReg() &&
4879 MRI->getRegClass(AMDGPU::VGPR_16RegClassID).contains(Op.getReg())) {
4880 bool VGPRSuffixIsHi = AMDGPU::isHi16Reg(Op.getReg(), *MRI);
4881 bool OpSelOpIsHi = ((OpSelOpValue & (1 << OpCount)) != 0);
4882 if (OpSelOpIsHi != VGPRSuffixIsHi)
4883 return false;
4884 }
4885 ++OpCount;
4886 }
4887
4888 return true;
4889}
4890
4891bool AMDGPUAsmParser::validateNeg(const MCInst &Inst, AMDGPU::OpName OpName) {
4892 assert(OpName == AMDGPU::OpName::neg_lo || OpName == AMDGPU::OpName::neg_hi);
4893
4894 const unsigned Opc = Inst.getOpcode();
4895
4896 // v_dot4 fp8/bf8 neg_lo/neg_hi not allowed on src0 and src1 (allowed on src2)
4897 // v_wmma iu4/iu8 neg_lo not allowed on src2 (allowed on src0, src1)
4898 // v_swmmac f16/bf16 neg_lo/neg_hi not allowed on src2 (allowed on src0, src1)
4899 // other wmma/swmmac instructions don't have neg_lo/neg_hi operand.
4900 if (!SIInstrFlags::isDOT(MII, Inst) && !SIInstrFlags::isWMMA(MII, Inst) &&
4901 !SIInstrFlags::isSWMMAC(MII, Inst))
4902 return true;
4903
4904 int NegIdx = AMDGPU::getNamedOperandIdx(Opc, OpName);
4905 if (NegIdx == -1)
4906 return true;
4907
4908 unsigned Neg = Inst.getOperand(NegIdx).getImm();
4909
4910 // Instructions that have neg_lo or neg_hi operand but neg modifier is allowed
4911 // on some src operands but not allowed on other.
4912 // It is convenient that such instructions don't have src_modifiers operand
4913 // for src operands that don't allow neg because they also don't allow opsel.
4914
4915 const AMDGPU::OpName SrcMods[3] = {AMDGPU::OpName::src0_modifiers,
4916 AMDGPU::OpName::src1_modifiers,
4917 AMDGPU::OpName::src2_modifiers};
4918
4919 for (unsigned i = 0; i < 3; ++i) {
4920 if (!AMDGPU::hasNamedOperand(Opc, SrcMods[i])) {
4921 if (Neg & (1 << i))
4922 return false;
4923 }
4924 }
4925
4926 return true;
4927}
4928
4929bool AMDGPUAsmParser::validateDPP(const MCInst &Inst,
4930 const OperandVector &Operands) {
4931 const unsigned Opc = Inst.getOpcode();
4932 int DppCtrlIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dpp_ctrl);
4933 if (DppCtrlIdx >= 0) {
4934 unsigned DppCtrl = Inst.getOperand(DppCtrlIdx).getImm();
4935
4936 if (!AMDGPU::isLegalDPALU_DPPControl(getSTI(), DppCtrl) &&
4937 getSTI().hasFeature(AMDGPU::FeatureDPALU_DPP) &&
4938 AMDGPU::isDPALU_DPP(MII.get(Opc), MII, getSTI())) {
4939 // DP ALU DPP is supported for row_newbcast only on GFX9* and row_share
4940 // only on GFX12.
4941 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyDppCtrl, Operands);
4942 Error(S, isGFX12() ? "DP ALU dpp only supports row_share"
4943 : "DP ALU dpp only supports row_newbcast");
4944 return false;
4945 }
4946 }
4947
4948 int Dpp8Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::dpp8);
4949 bool IsDPP = DppCtrlIdx >= 0 || Dpp8Idx >= 0;
4950
4951 if (IsDPP && !hasDPPSrc1SGPR(getSTI())) {
4952 int Src1Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src1);
4953 if (Src1Idx >= 0) {
4954 const MCOperand &Src1 = Inst.getOperand(Src1Idx);
4955 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
4956 if (Src1.isReg() && isSGPR(mc2PseudoReg(Src1.getReg()), TRI)) {
4957 Error(getOperandLoc(Operands, Src1Idx),
4958 "invalid operand for instruction");
4959 return false;
4960 }
4961 if (Src1.isImm()) {
4962 Error(getInstLoc(Operands),
4963 "src1 immediate operand invalid for instruction");
4964 return false;
4965 }
4966 }
4967 }
4968
4969 return true;
4970}
4971
4972// Check if VCC register matches wavefront size
4973bool AMDGPUAsmParser::validateVccOperand(MCRegister Reg) const {
4974 return (Reg == AMDGPU::VCC && isWave64()) ||
4975 (Reg == AMDGPU::VCC_LO && isWave32());
4976}
4977
4978// One unique literal can be used. VOP3 literal is only allowed in GFX10+
4979bool AMDGPUAsmParser::validateVOPLiteral(const MCInst &Inst,
4980 const OperandVector &Operands) {
4981 unsigned Opcode = Inst.getOpcode();
4982 const MCInstrDesc &Desc = MII.get(Opcode);
4983 bool HasMandatoryLiteral = getNamedOperandIdx(Opcode, OpName::imm) != -1;
4984 if (!SIInstrFlags::isVOP3Like(Desc) && !HasMandatoryLiteral &&
4985 !isVOPD(Opcode))
4986 return true;
4987
4988 OperandIndices OpIndices = getSrcOperandIndices(Opcode, HasMandatoryLiteral);
4989
4990 std::optional<unsigned> LiteralOpIdx;
4991 std::optional<uint64_t> LiteralValue;
4992
4993 for (int OpIdx : OpIndices) {
4994 if (OpIdx == -1)
4995 continue;
4996
4997 const MCOperand &MO = Inst.getOperand(OpIdx);
4998 if (!MO.isImm() && !MO.isExpr())
4999 continue;
5000 if (!isSISrcOperand(Desc, OpIdx))
5001 continue;
5002
5003 std::optional<int64_t> Imm;
5004 if (MO.isImm())
5005 Imm = MO.getImm();
5006 else if (MO.isExpr() && isLitExpr(MO.getExpr()))
5007 Imm = getLitValue(MO.getExpr());
5008
5009 bool IsAnotherLiteral = false;
5010 bool IsForcedLit = findMCOperand(Operands, OpIdx).isForcedLit();
5011 bool IsForcedLit64 = findMCOperand(Operands, OpIdx).isForcedLit64();
5012 if (!Imm.has_value()) {
5013 // Literal value not known, so we conservately assume it's different.
5014 IsAnotherLiteral = true;
5015 } else if (IsForcedLit || IsForcedLit64 || !isInlineConstant(Inst, OpIdx)) {
5016 uint64_t Value = *Imm;
5017 bool IsForcedFP64 =
5018 Desc.operands()[OpIdx].OperandType == AMDGPU::OPERAND_KIMM64 ||
5019 (Desc.operands()[OpIdx].OperandType == AMDGPU::OPERAND_REG_IMM_FP64 &&
5020 HasMandatoryLiteral);
5021 AMDGPU::OperandType OpTy =
5022 static_cast<AMDGPU::OperandType>(Desc.operands()[OpIdx].OperandType);
5023 bool IsFP64 =
5024 (IsForcedFP64 || (AMDGPU::isSISrcFPOperand(Desc, OpIdx) &&
5026 AMDGPU::getOperandSize(Desc.operands()[OpIdx]) == 8;
5027 bool IsValid32Op =
5028 IsForcedLit || AMDGPU::isValid32BitLiteral(Value, IsFP64);
5029
5030 if (((!IsValid32Op && !isInt<32>(Value) && !isUInt<32>(Value) &&
5031 !IsForcedFP64) ||
5032 (IsForcedLit64 && !HasMandatoryLiteral)) &&
5033 (!has64BitLiterals() || Desc.getSize() != 4)) {
5034 Error(getOperandLoc(Operands, OpIdx),
5035 "invalid operand for instruction");
5036 return false;
5037 }
5038
5039 // Only src0 can use lit64 in VOP* encoding.
5040 if (!IsForcedFP64 && (IsForcedLit64 || !IsValid32Op) &&
5041 OpIdx != getNamedOperandIdx(Opcode, OpName::src0)) {
5042 Error(getOperandLoc(Operands, OpIdx),
5043 "invalid operand for instruction");
5044 return false;
5045 }
5046
5047 // Compare values using the word encoded by a 32-bit literal.
5048 if (IsValid32Op && !IsForcedFP64 && !IsForcedLit64) {
5049 Value = static_cast<uint32_t>(
5050 AMDGPU::encode32BitLiteral(Value, OpTy, IsForcedLit));
5051 }
5052
5053 IsAnotherLiteral = !LiteralValue || *LiteralValue != Value;
5055 }
5056
5057 if (IsAnotherLiteral && !HasMandatoryLiteral &&
5058 !getFeatureBits()[FeatureVOP3Literal]) {
5059 Error(getOperandLoc(Operands, OpIdx),
5060 "literal operands are not supported");
5061 return false;
5062 }
5063
5064 if (LiteralOpIdx && IsAnotherLiteral) {
5065 Error(getLaterLoc(getOperandLoc(Operands, OpIdx),
5066 getOperandLoc(Operands, *LiteralOpIdx)),
5067 "only one unique literal operand is allowed");
5068 return false;
5069 }
5070
5071 if (IsAnotherLiteral)
5072 LiteralOpIdx = OpIdx;
5073 }
5074
5075 return true;
5076}
5077
5078// Returns -1 if not a register, 0 if VGPR and 1 if AGPR.
5079static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name,
5080 const MCRegisterInfo *MRI) {
5081 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.getOpcode(), Name);
5082 if (OpIdx < 0)
5083 return -1;
5084
5085 const MCOperand &Op = Inst.getOperand(OpIdx);
5086 if (!Op.isReg())
5087 return -1;
5088
5089 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5090 auto Reg = Sub ? Sub : Op.getReg();
5091 const MCRegisterClass &AGPR32 = MRI->getRegClass(AMDGPU::AGPR_32RegClassID);
5092 return AGPR32.contains(Reg) ? 1 : 0;
5093}
5094
5095bool AMDGPUAsmParser::validateAGPRLdSt(const MCInst &Inst) const {
5096 if (!SIInstrFlags::isFLAT(MII, Inst) && !SIInstrFlags::isBuffer(MII, Inst) &&
5097 !SIInstrFlags::isMIMG(MII, Inst) && !SIInstrFlags::isDS(MII, Inst))
5098 return true;
5099
5100 AMDGPU::OpName DataName = SIInstrFlags::isDS(MII, Inst)
5101 ? AMDGPU::OpName::data0
5102 : AMDGPU::OpName::vdata;
5103
5104 const MCRegisterInfo *MRI = getMRI();
5105 int DstAreg = IsAGPROperand(Inst, AMDGPU::OpName::vdst, MRI);
5106 int DataAreg = IsAGPROperand(Inst, DataName, MRI);
5107
5108 if (SIInstrFlags::isDS(MII, Inst) && DataAreg >= 0) {
5109 int Data2Areg = IsAGPROperand(Inst, AMDGPU::OpName::data1, MRI);
5110 if (Data2Areg >= 0 && Data2Areg != DataAreg)
5111 return false;
5112 }
5113
5114 auto FB = getFeatureBits();
5115 if (FB[AMDGPU::FeatureGFX90AInsts]) {
5116 if (DataAreg < 0 || DstAreg < 0)
5117 return true;
5118 return DstAreg == DataAreg;
5119 }
5120
5121 return DstAreg < 1 && DataAreg < 1;
5122}
5123
5124bool AMDGPUAsmParser::validateVGPRAlign(const MCInst &Inst) const {
5125 auto FB = getFeatureBits();
5126 if (!FB[AMDGPU::FeatureRequiresAlignedVGPRs])
5127 return true;
5128
5129 unsigned Opc = Inst.getOpcode();
5130 const MCRegisterInfo *MRI = getMRI();
5131 // DS_READ_B96_TR_B6 is the only DS instruction in GFX950, that allows
5132 // unaligned VGPR. All others only allow even aligned VGPRs.
5133 if (FB[AMDGPU::FeatureGFX90AInsts] && Opc == AMDGPU::DS_READ_B96_TR_B6_vi)
5134 return true;
5135
5136 if (FB[AMDGPU::FeatureGFX1250Insts]) {
5137 switch (Opc) {
5138 default:
5139 break;
5140 case AMDGPU::DS_LOAD_TR6_B96:
5141 case AMDGPU::DS_LOAD_TR6_B96_gfx12:
5142 // DS_LOAD_TR6_B96 is the only DS instruction in GFX1250, that
5143 // allows unaligned VGPR. All others only allow even aligned VGPRs.
5144 return true;
5145 case AMDGPU::GLOBAL_LOAD_TR6_B96:
5146 case AMDGPU::GLOBAL_LOAD_TR6_B96_gfx1250: {
5147 // GLOBAL_LOAD_TR6_B96 is the only GLOBAL instruction in GFX1250, that
5148 // allows unaligned VGPR for vdst, but other operands still only allow
5149 // even aligned VGPRs.
5150 int VAddrIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vaddr);
5151 if (VAddrIdx != -1) {
5152 const MCOperand &Op = Inst.getOperand(VAddrIdx);
5153 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5154 if ((Sub - AMDGPU::VGPR0) & 1)
5155 return false;
5156 }
5157 return true;
5158 }
5159 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR:
5160 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR_gfx1250:
5161 return true;
5162 }
5163 }
5164
5165 const MCRegisterClass &VGPR32 = MRI->getRegClass(AMDGPU::VGPR_32RegClassID);
5166 const MCRegisterClass &AGPR32 = MRI->getRegClass(AMDGPU::AGPR_32RegClassID);
5167 for (unsigned I = 0, E = Inst.getNumOperands(); I != E; ++I) {
5168 const MCOperand &Op = Inst.getOperand(I);
5169 if (!Op.isReg())
5170 continue;
5171
5172 MCRegister Sub = MRI->getSubReg(Op.getReg(), AMDGPU::sub0);
5173 if (!Sub)
5174 continue;
5175
5176 if (VGPR32.contains(Sub) && ((Sub - AMDGPU::VGPR0) & 1))
5177 return false;
5178 if (AGPR32.contains(Sub) && ((Sub - AMDGPU::AGPR0) & 1))
5179 return false;
5180 }
5181
5182 return true;
5183}
5184
5185SMLoc AMDGPUAsmParser::getBLGPLoc(const OperandVector &Operands) const {
5186 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
5187 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
5188 if (Op.isBLGP())
5189 return Op.getStartLoc();
5190 }
5191 return SMLoc();
5192}
5193
5194bool AMDGPUAsmParser::validateBLGP(const MCInst &Inst,
5195 const OperandVector &Operands) {
5196 unsigned Opc = Inst.getOpcode();
5197 int BlgpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
5198 if (BlgpIdx == -1)
5199 return true;
5200 SMLoc BLGPLoc = getBLGPLoc(Operands);
5201 if (!BLGPLoc.isValid())
5202 return true;
5203 bool IsNeg = StringRef(BLGPLoc.getPointer()).starts_with("neg:");
5204 auto FB = getFeatureBits();
5205 bool UsesNeg = false;
5206 if (FB[AMDGPU::FeatureGFX940Insts]) {
5207 switch (Opc) {
5208 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_acd:
5209 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_vcd:
5210 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_acd:
5211 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_vcd:
5212 UsesNeg = true;
5213 }
5214 }
5215
5216 if (IsNeg == UsesNeg)
5217 return true;
5218
5219 Error(BLGPLoc, UsesNeg ? "invalid modifier: blgp is not supported"
5220 : "invalid modifier: neg is not supported");
5221
5222 return false;
5223}
5224
5225bool AMDGPUAsmParser::validateWaitCnt(const MCInst &Inst,
5226 const OperandVector &Operands) {
5227 if (!isGFX11Plus())
5228 return true;
5229
5230 unsigned Opc = Inst.getOpcode();
5231 if (Opc != AMDGPU::S_WAITCNT_EXPCNT_gfx11 &&
5232 Opc != AMDGPU::S_WAITCNT_LGKMCNT_gfx11 &&
5233 Opc != AMDGPU::S_WAITCNT_VMCNT_gfx11 &&
5234 Opc != AMDGPU::S_WAITCNT_VSCNT_gfx11)
5235 return true;
5236
5237 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::sdst);
5238 assert(Src0Idx >= 0 && Inst.getOperand(Src0Idx).isReg());
5239 auto Reg = mc2PseudoReg(Inst.getOperand(Src0Idx).getReg());
5240 if (Reg == AMDGPU::SGPR_NULL)
5241 return true;
5242
5243 Error(getOperandLoc(Operands, Src0Idx), "src0 must be null");
5244 return false;
5245}
5246
5247bool AMDGPUAsmParser::validateDS(const MCInst &Inst,
5248 const OperandVector &Operands) {
5249 if (!SIInstrFlags::isDS(MII, Inst))
5250 return true;
5251 if (SIInstrFlags::isGWS(MII, Inst))
5252 return validateGWS(Inst, Operands);
5253 // Only validate GDS for non-GWS instructions.
5254 if (hasGDS())
5255 return true;
5256 int GDSIdx =
5257 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::gds);
5258 if (GDSIdx < 0)
5259 return true;
5260 unsigned GDS = Inst.getOperand(GDSIdx).getImm();
5261 if (GDS) {
5262 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyGDS, Operands);
5263 Error(S, "gds modifier is not supported on this GPU");
5264 return false;
5265 }
5266 return true;
5267}
5268
5269// gfx90a has an undocumented limitation:
5270// DS_GWS opcodes must use even aligned registers.
5271bool AMDGPUAsmParser::validateGWS(const MCInst &Inst,
5272 const OperandVector &Operands) {
5273 if (!getFeatureBits()[AMDGPU::FeatureGFX90AInsts])
5274 return true;
5275
5276 int Opc = Inst.getOpcode();
5277 if (Opc != AMDGPU::DS_GWS_INIT_vi && Opc != AMDGPU::DS_GWS_BARRIER_vi &&
5278 Opc != AMDGPU::DS_GWS_SEMA_BR_vi)
5279 return true;
5280
5281 const MCRegisterInfo *MRI = getMRI();
5282 const MCRegisterClass &VGPR32 = MRI->getRegClass(AMDGPU::VGPR_32RegClassID);
5283 int Data0Pos =
5284 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::data0);
5285 assert(Data0Pos != -1);
5286 auto Reg = Inst.getOperand(Data0Pos).getReg();
5287 auto RegIdx = Reg - (VGPR32.contains(Reg) ? AMDGPU::VGPR0 : AMDGPU::AGPR0);
5288 if (RegIdx & 1) {
5289 Error(getOperandLoc(Operands, Data0Pos), "vgpr must be even aligned");
5290 return false;
5291 }
5292
5293 return true;
5294}
5295
5296bool AMDGPUAsmParser::validateCoherencyBits(const MCInst &Inst,
5297 const OperandVector &Operands,
5298 SMLoc IDLoc) {
5299 int CPolPos =
5300 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::cpol);
5301 if (CPolPos == -1)
5302 return true;
5303
5304 unsigned CPol = Inst.getOperand(CPolPos).getImm();
5305
5306 if (!isGFX1250Plus()) {
5307 if (CPol & CPol::SCAL) {
5308 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5309 StringRef CStr(S.getPointer());
5310 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scale_offset")]);
5311 Error(S, "scale_offset is not supported on this GPU");
5312 }
5313 if (CPol & CPol::NV) {
5314 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5315 StringRef CStr(S.getPointer());
5316 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("nv")]);
5317 Error(S, "nv is not supported on this GPU");
5318 }
5319 }
5320
5321 if ((CPol & CPol::SCAL) && !supportsScaleOffset(MII, Inst.getOpcode())) {
5322 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5323 StringRef CStr(S.getPointer());
5324 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scale_offset")]);
5325 Error(S, "scale_offset is not supported for this instruction");
5326 }
5327
5328 if (isGFX12Plus())
5329 return validateTHAndScopeBits(Inst, Operands, CPol);
5330
5331 if (SIInstrFlags::isSMRD(MII, Inst)) {
5332 if (CPol && (isSI() || isCI())) {
5333 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5334 Error(S, "cache policy is not supported for SMRD instructions");
5335 return false;
5336 }
5337 if (CPol & ~(AMDGPU::CPol::GLC | AMDGPU::CPol::DLC)) {
5338 Error(IDLoc, "invalid cache policy for SMEM instruction");
5339 return false;
5340 }
5341 }
5342
5343 if (isGFX90A() && !isGFX940() && (CPol & CPol::SCC)) {
5344 if (!SIInstrFlags::isVMEM(MII, Inst)) {
5345 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5346 StringRef CStr(S.getPointer());
5347 S = SMLoc::getFromPointer(&CStr.data()[CStr.find("scc")]);
5348 Error(S,
5349 "scc modifier is not supported for this instruction on this GPU");
5350 return false;
5351 }
5352 }
5353
5354 if (!SIInstrFlags::isAtomic(MII, Inst))
5355 return true;
5356
5357 if (SIInstrFlags::isAtomicRet(MII, Inst)) {
5358 if (!SIInstrFlags::isMIMG(MII, Inst) && !(CPol & CPol::GLC)) {
5359 Error(IDLoc, isGFX940() ? "instruction must use sc0"
5360 : "instruction must use glc");
5361 return false;
5362 }
5363 } else {
5364 if (CPol & CPol::GLC) {
5365 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5366 StringRef CStr(S.getPointer());
5368 &CStr.data()[CStr.find(isGFX940() ? "sc0" : "glc")]);
5369 Error(S, isGFX940() ? "instruction must not use sc0"
5370 : "instruction must not use glc");
5371 return false;
5372 }
5373 }
5374
5375 return true;
5376}
5377
5378bool AMDGPUAsmParser::validateTHAndScopeBits(const MCInst &Inst,
5379 const OperandVector &Operands,
5380 const unsigned CPol) {
5381 const unsigned TH = CPol & AMDGPU::CPol::TH;
5382 const unsigned Scope = CPol & AMDGPU::CPol::SCOPE;
5383
5384 auto PrintError = [&](StringRef Msg) {
5385 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol, Operands);
5386 Error(S, Msg);
5387 return false;
5388 };
5389
5390 if ((TH & AMDGPU::CPol::TH_ATOMIC_RETURN) &&
5391 SIInstrFlags::isAtomicNoRet(MII, Inst))
5392 return PrintError("th:TH_ATOMIC_RETURN requires a destination operand");
5393
5394 if (SIInstrFlags::isAtomicRet(MII, Inst) &&
5395 (SIInstrFlags::isFLAT(MII, Inst) || SIInstrFlags::isMUBUF(MII, Inst)) &&
5397 return PrintError("instruction must use th:TH_ATOMIC_RETURN");
5398
5399 if (TH == 0)
5400 return true;
5401
5402 if (SIInstrFlags::isSMRD(MII, Inst) &&
5403 ((TH == AMDGPU::CPol::TH_NT_RT) || (TH == AMDGPU::CPol::TH_RT_NT) ||
5404 (TH == AMDGPU::CPol::TH_NT_HT)))
5405 return PrintError("invalid th value for SMEM instruction");
5406
5407 if (TH == AMDGPU::CPol::TH_BYPASS) {
5408 if ((Scope != AMDGPU::CPol::SCOPE_SYS &&
5410 (Scope == AMDGPU::CPol::SCOPE_SYS &&
5412 return PrintError("scope and th combination is not valid");
5413 }
5414
5415 unsigned THType = AMDGPU::getTemporalHintType(MII.get(Inst.getOpcode()));
5416 if (THType == AMDGPU::CPol::TH_TYPE_ATOMIC) {
5417 if (!(CPol & AMDGPU::CPol::TH_TYPE_ATOMIC))
5418 return PrintError("invalid th value for atomic instructions");
5419 } else if (THType == AMDGPU::CPol::TH_TYPE_STORE) {
5420 if (!(CPol & AMDGPU::CPol::TH_TYPE_STORE))
5421 return PrintError("invalid th value for store instructions");
5422 } else {
5423 if (!(CPol & AMDGPU::CPol::TH_TYPE_LOAD))
5424 return PrintError("invalid th value for load instructions");
5425 }
5426
5427 return true;
5428}
5429
5430bool AMDGPUAsmParser::validateTFE(const MCInst &Inst,
5431 const OperandVector &Operands) {
5432 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
5433 if (Desc.mayStore() && SIInstrFlags::isBuffer(Desc)) {
5434 SMLoc Loc = getImmLoc(AMDGPUOperand::ImmTyTFE, Operands);
5435 if (Loc != getInstLoc(Operands)) {
5436 Error(Loc, "TFE modifier has no meaning for store instructions");
5437 return false;
5438 }
5439 }
5440
5441 return true;
5442}
5443
5444bool AMDGPUAsmParser::validateWMMA(const MCInst &Inst,
5445 const OperandVector &Operands) {
5446 unsigned Opc = Inst.getOpcode();
5447 const MCRegisterInfo *TRI = getContext().getRegisterInfo();
5448 const MCInstrDesc &Desc = MII.get(Opc);
5449
5450 int AFmtIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_fmt);
5451 if (AFmtIdx == -1)
5452 return true;
5453 unsigned AFmt = Inst.getOperand(AFmtIdx).getImm();
5454 int BFmtIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_fmt);
5455 unsigned BFmt = Inst.getOperand(BFmtIdx).getImm();
5456
5457 auto validateFmt = [&](unsigned Fmt, AMDGPU::OpName SrcOp) -> bool {
5458 int SrcIdx = AMDGPU::getNamedOperandIdx(Opc, SrcOp);
5459 unsigned RegSize =
5460 TRI->getRegClass(MII.getOpRegClassID(Desc.operands()[SrcIdx], HwMode))
5461 .getSizeInBits();
5462
5464 return true;
5465
5466 Error(getOperandLoc(Operands, SrcIdx),
5467 "wrong register tuple size for " +
5468 Twine(WMMAMods::ModMatrixFmt[Fmt]));
5469 return false;
5470 };
5471
5472 if (!validateFmt(AFmt, AMDGPU::OpName::src0) ||
5473 !validateFmt(BFmt, AMDGPU::OpName::src1))
5474 return false;
5475
5476 int AScaleIdx =
5477 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale_fmt);
5478 if (AScaleIdx == -1)
5479 return true;
5480 unsigned AScale = Inst.getOperand(AScaleIdx).getImm();
5481 int BScaleIdx =
5482 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale_fmt);
5483 unsigned BScale = Inst.getOperand(BScaleIdx).getImm();
5484 if (!isValidWMMAScaleFmtCombination(AFmt, AScale, BFmt, BScale)) {
5485 Error(getImmLoc(AMDGPUOperand::ImmTyMatrixAFMT, Operands),
5486 "invalid matrix and scale format combination");
5487 return false;
5488 }
5489
5490 return true;
5491}
5492
5493bool AMDGPUAsmParser::validateMonitorSleep(const MCInst &Inst,
5494 const OperandVector &Operands) {
5495 unsigned Opc = Inst.getOpcode();
5496 if (Opc != AMDGPU::S_MONITOR_SLEEP_gfx12 ||
5497 !getSTI().hasFeature(AMDGPU::FeatureNoSleepForever))
5498 return true;
5499
5500 int ImmIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::simm16);
5501 if (Inst.getOperand(ImmIdx).getImm() & 0x8000) {
5502 Error(getOperandLoc(Operands, ImmIdx),
5503 "sleep forever is unsuported on the target");
5504 return false;
5505 }
5506
5507 return true;
5508}
5509
5510bool AMDGPUAsmParser::validateClusterBarrierIsFirst(
5511 const MCInst &Inst, const OperandVector &Operands) {
5512 unsigned Opc = Inst.getOpcode();
5513 if (Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx12 &&
5514 Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx13)
5515 return true;
5516
5517 int Src0Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0);
5518 int BarrierID = Inst.getOperand(Src0Idx).getImm();
5519 if (BarrierID != AMDGPU::Barrier::CLUSTER)
5520 return true;
5521
5522 Error(
5523 getOperandLoc(Operands, Src0Idx),
5524 "s_barrier_signal_isfirst does not support user_cluster_barrier_id (-3)");
5525 return false;
5526}
5527
5528bool AMDGPUAsmParser::validateScaleSel(const MCInst &Inst,
5529 const OperandVector &Operands) {
5530 unsigned Opc = Inst.getOpcode();
5531 int ScaleSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::scale_sel);
5532 if (ScaleSelIdx == -1)
5533 return true;
5534 int MaxSel = 0;
5535 switch (Opc) {
5536 case AMDGPU::V_CVT_SCALE_PK16_F16_FP6_e64_gfx1250:
5537 case AMDGPU::V_CVT_SCALE_PK16_BF16_FP6_e64_gfx1250:
5538 case AMDGPU::V_CVT_SCALE_PK16_F16_BF6_e64_gfx1250:
5539 case AMDGPU::V_CVT_SCALE_PK16_BF16_BF6_e64_gfx1250:
5540 case AMDGPU::V_CVT_SCALE_PK16_F32_FP6_e64_gfx1250:
5541 case AMDGPU::V_CVT_SCALE_PK16_F32_BF6_e64_gfx1250:
5542 case AMDGPU::V_CVT_SCALE_PK8_F16_FP4_e64_gfx1250:
5543 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP4_e64_gfx1250:
5544 case AMDGPU::V_CVT_SCALE_PK8_F32_FP4_e64_gfx1250:
5545 MaxSel = 4;
5546 break;
5547 case AMDGPU::V_CVT_SCALE_PK8_F16_FP8_e64_gfx1250:
5548 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP8_e64_gfx1250:
5549 case AMDGPU::V_CVT_SCALE_PK8_F16_BF8_e64_gfx1250:
5550 case AMDGPU::V_CVT_SCALE_PK8_BF16_BF8_e64_gfx1250:
5551 case AMDGPU::V_CVT_SCALE_PK8_F32_FP8_e64_gfx1250:
5552 case AMDGPU::V_CVT_SCALE_PK8_F32_BF8_e64_gfx1250:
5553 MaxSel = 8;
5554 break;
5555 default:
5556 return true;
5557 }
5558
5559 if (getSTI().hasFeature(AMDGPU::FeatureBlock16ConversionScaleInsts))
5560 MaxSel *= 2;
5561
5562 int ScaleSel = Inst.getOperand(ScaleSelIdx).getImm();
5563 if (ScaleSel < MaxSel)
5564 return true;
5565
5566 Error(getOperandLoc(Operands, ScaleSelIdx),
5567 "scale_sel maximum supported value is " + Twine(MaxSel - 1));
5568 return false;
5569}
5570
5571bool AMDGPUAsmParser::validateInstruction(const MCInst &Inst, SMLoc IDLoc,
5572 const OperandVector &Operands) {
5573 if (!validateLdsDirect(Inst, Operands))
5574 return false;
5575 if (!validateTrue16OpSel(Inst)) {
5576 Error(getImmLoc(AMDGPUOperand::ImmTyOpSel, Operands),
5577 "op_sel operand conflicts with 16-bit operand suffix");
5578 return false;
5579 }
5580 if (!validateSOPLiteral(Inst, Operands))
5581 return false;
5582 if (!validateVOPLiteral(Inst, Operands)) {
5583 return false;
5584 }
5585 if (!validateConstantBusLimitations(Inst, Operands)) {
5586 return false;
5587 }
5588 if (!validateVOPD(Inst, Operands)) {
5589 return false;
5590 }
5591 if (!validateIntClampSupported(Inst)) {
5592 Error(getImmLoc(AMDGPUOperand::ImmTyClamp, Operands),
5593 "integer clamping is not supported on this GPU");
5594 return false;
5595 }
5596 if (!validateOpSel(Inst)) {
5597 Error(getImmLoc(AMDGPUOperand::ImmTyOpSel, Operands),
5598 "invalid op_sel operand");
5599 return false;
5600 }
5601 if (!validateNeg(Inst, AMDGPU::OpName::neg_lo)) {
5602 Error(getImmLoc(AMDGPUOperand::ImmTyNegLo, Operands),
5603 "invalid neg_lo operand");
5604 return false;
5605 }
5606 if (!validateNeg(Inst, AMDGPU::OpName::neg_hi)) {
5607 Error(getImmLoc(AMDGPUOperand::ImmTyNegHi, Operands),
5608 "invalid neg_hi operand");
5609 return false;
5610 }
5611 if (!validateDPP(Inst, Operands)) {
5612 return false;
5613 }
5614 // For MUBUF/MTBUF d16 is a part of opcode, so there is nothing to validate.
5615 if (!validateMIMGD16(Inst)) {
5616 Error(getImmLoc(AMDGPUOperand::ImmTyD16, Operands),
5617 "d16 modifier is not supported on this GPU");
5618 return false;
5619 }
5620 if (!validateMIMGDim(Inst, Operands)) {
5621 Error(IDLoc, "missing dim operand");
5622 return false;
5623 }
5624 if (!validateTensorR128(Inst)) {
5625 Error(getImmLoc(AMDGPUOperand::ImmTyD16, Operands),
5626 "instruction must set modifier r128=0");
5627 return false;
5628 }
5629 if (!validateMIMGMSAA(Inst)) {
5630 Error(getImmLoc(AMDGPUOperand::ImmTyDim, Operands),
5631 "invalid dim; must be MSAA type");
5632 return false;
5633 }
5634 if (!validateMIMGDataSize(Inst, IDLoc)) {
5635 return false;
5636 }
5637 if (!validateMIMGAddrSize(Inst, IDLoc))
5638 return false;
5639 if (!validateMIMGAtomicDMask(Inst)) {
5640 Error(getImmLoc(AMDGPUOperand::ImmTyDMask, Operands),
5641 "invalid atomic image dmask");
5642 return false;
5643 }
5644 if (!validateMIMGGatherDMask(Inst)) {
5645 Error(getImmLoc(AMDGPUOperand::ImmTyDMask, Operands),
5646 "invalid image_gather dmask: only one bit must be set");
5647 return false;
5648 }
5649 if (!validateMovrels(Inst, Operands)) {
5650 return false;
5651 }
5652 if (!validateOffset(Inst, Operands)) {
5653 return false;
5654 }
5655 if (!validateBF16InlineConst(Inst, Operands)) {
5656 return false;
5657 }
5658 if (!validateMAIAccWrite(Inst, Operands)) {
5659 return false;
5660 }
5661 if (!validateMAISrc2(Inst, Operands)) {
5662 return false;
5663 }
5664 if (!validateMFMA(Inst, Operands)) {
5665 return false;
5666 }
5667 if (!validateCoherencyBits(Inst, Operands, IDLoc)) {
5668 return false;
5669 }
5670
5671 if (!validateAGPRLdSt(Inst)) {
5672 Error(
5673 IDLoc,
5674 getFeatureBits()[AMDGPU::FeatureGFX90AInsts]
5675 ? "invalid register class: data and dst should be all VGPR or AGPR"
5676 : "invalid register class: agpr loads and stores not supported on "
5677 "this GPU");
5678 return false;
5679 }
5680 if (!validateVGPRAlign(Inst)) {
5681 Error(IDLoc, "invalid register class: vgpr tuples must be 64 bit aligned");
5682 return false;
5683 }
5684 if (!validateDS(Inst, Operands)) {
5685 return false;
5686 }
5687
5688 if (!validateBLGP(Inst, Operands)) {
5689 return false;
5690 }
5691
5692 if (!validateDivScale(Inst)) {
5693 Error(IDLoc, "ABS not allowed in VOP3B instructions");
5694 return false;
5695 }
5696 if (!validateWaitCnt(Inst, Operands)) {
5697 return false;
5698 }
5699 if (!validateTFE(Inst, Operands)) {
5700 return false;
5701 }
5702 if (!validateWMMA(Inst, Operands)) {
5703 return false;
5704 }
5705 if (!validateMonitorSleep(Inst, Operands)) {
5706 return false;
5707 }
5708 if (!validateClusterBarrierIsFirst(Inst, Operands)) {
5709 return false;
5710 }
5711 if (!validateScaleSel(Inst, Operands)) {
5712 return false;
5713 }
5714
5715 return true;
5716}
5717
5719 const FeatureBitset &FBS,
5720 unsigned VariantID = 0);
5721
5722static bool AMDGPUCheckMnemonic(StringRef Mnemonic,
5723 const FeatureBitset &AvailableFeatures,
5724 unsigned VariantID);
5725
5726bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5727 const FeatureBitset &FBS) {
5728 return isSupportedMnemo(Mnemo, FBS, getAllVariants());
5729}
5730
5731bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5732 const FeatureBitset &FBS,
5733 ArrayRef<unsigned> Variants) {
5734 for (auto Variant : Variants) {
5735 if (AMDGPUCheckMnemonic(Mnemo, FBS, Variant))
5736 return true;
5737 }
5738
5739 return false;
5740}
5741
5742bool AMDGPUAsmParser::checkUnsupportedInstruction(StringRef Mnemo,
5743 SMLoc IDLoc) {
5744 FeatureBitset FBS = ComputeAvailableFeatures(getFeatureBits());
5745
5746 // Check if requested instruction variant is supported.
5747 if (isSupportedMnemo(Mnemo, FBS, getMatchedVariants()))
5748 return false;
5749
5750 // This instruction is not supported.
5751 // Clear any other pending errors because they are no longer relevant.
5752 getParser().clearPendingErrors();
5753
5754 // Requested instruction variant is not supported.
5755 // Check if any other variants are supported.
5756 StringRef VariantName = getMatchedVariantName();
5757 if (!VariantName.empty() && isSupportedMnemo(Mnemo, FBS)) {
5758 return Error(IDLoc, Twine(VariantName,
5759 " variant of this instruction is not supported"));
5760 }
5761
5762 // Check if this instruction may be used with a different wavesize.
5763 if (isGFX10Plus() && getFeatureBits()[AMDGPU::FeatureWavefrontSize64] &&
5764 !getFeatureBits()[AMDGPU::FeatureWavefrontSize32]) {
5765 // FIXME: Use getAvailableFeatures, and do not manually recompute
5766 FeatureBitset FeaturesWS32 = getFeatureBits();
5767 FeaturesWS32.flip(AMDGPU::FeatureWavefrontSize64)
5768 .flip(AMDGPU::FeatureWavefrontSize32);
5769 FeatureBitset AvailableFeaturesWS32 =
5770 ComputeAvailableFeatures(FeaturesWS32);
5771
5772 if (isSupportedMnemo(Mnemo, AvailableFeaturesWS32, getMatchedVariants()))
5773 return Error(IDLoc, "instruction requires wavesize=32");
5774 }
5775
5776 // Finally check if this instruction is supported on any other GPU.
5777 if (isSupportedMnemo(Mnemo, FeatureBitset().set())) {
5778 return Error(IDLoc, "instruction not supported on this GPU (" +
5779 getSTI().getCPU() + ")" + ": " + Mnemo);
5780 }
5781
5782 // Instruction not supported on any GPU. Probably a typo.
5783 std::string Suggestion = AMDGPUMnemonicSpellCheck(Mnemo, FBS);
5784 return Error(IDLoc, "invalid instruction" + Suggestion);
5785}
5786
5788 uint64_t InvalidOprIdx) {
5789 assert(InvalidOprIdx < Operands.size());
5790 const auto &Op = ((AMDGPUOperand &)*Operands[InvalidOprIdx]);
5791 if (Op.isToken() && InvalidOprIdx > 1) {
5792 const auto &PrevOp = ((AMDGPUOperand &)*Operands[InvalidOprIdx - 1]);
5793 return PrevOp.isToken() && PrevOp.getToken() == "::";
5794 }
5795 return false;
5796}
5797
5798bool AMDGPUAsmParser::matchAndEmitInstruction(SMLoc IDLoc, unsigned &Opcode,
5800 MCStreamer &Out,
5801 uint64_t &ErrorInfo,
5802 bool MatchingInlineAsm) {
5803 MCInst Inst;
5804 Inst.setLoc(IDLoc);
5805 unsigned Result = Match_Success;
5806
5807 // Order match statuses from least to most specific and keep the most
5808 // specific one:
5809 // Match_MnemonicFail < Match_InvalidOperand < Match_MissingFeature
5810 auto atLeastAsSpecific = [](unsigned New, unsigned Cur) {
5811 auto rank = [](unsigned M) {
5812 return M == Match_MnemonicFail ? 1
5813 : M == Match_InvalidOperand ? 2
5814 : M == Match_MissingFeature ? 3
5815 : 0; // Match_Success sentinel
5816 };
5817 return rank(New) >= rank(Cur);
5818 };
5819
5820 for (auto Variant : getMatchedVariants()) {
5821 uint64_t EI;
5822 auto R =
5823 MatchInstructionImpl(Operands, Inst, EI, MatchingInlineAsm, Variant);
5824 if (R == Match_Success || atLeastAsSpecific(R, Result)) {
5825 Result = R;
5826 ErrorInfo = EI;
5827 }
5828 if (R == Match_Success)
5829 break;
5830 }
5831
5832 if (Result == Match_Success) {
5833 if (!validateInstruction(Inst, IDLoc, Operands)) {
5834 return true;
5835 }
5836 emitTargetDirective();
5837 Out.emitInstruction(Inst, getSTI());
5838 // Record for kernel prologue checking.
5839 OpcodeStream.push_back(Inst.getOpcode());
5840 return false;
5841 }
5842
5843 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
5844 if (checkUnsupportedInstruction(Mnemo, IDLoc)) {
5845 return true;
5846 }
5847
5848 switch (Result) {
5849 default:
5850 break;
5851 case Match_MissingFeature:
5852 // It has been verified that the specified instruction
5853 // mnemonic is valid. A match was found but it requires
5854 // features which are not supported on this GPU.
5855 return Error(IDLoc, "operands are not valid for this GPU or mode");
5856
5857 case Match_InvalidOperand: {
5858 SMLoc ErrorLoc = IDLoc;
5859 if (ErrorInfo != ~0ULL) {
5860 if (ErrorInfo >= Operands.size()) {
5861 return Error(IDLoc, "too few operands for instruction");
5862 }
5863 AMDGPUOperand &ErrorOp = (AMDGPUOperand &)*Operands[ErrorInfo];
5864 ErrorLoc = ErrorOp.getStartLoc();
5865 if (ErrorLoc == SMLoc())
5866 ErrorLoc = IDLoc;
5867
5868 if (isInvalidVOPDY(Operands, ErrorInfo))
5869 return Error(ErrorLoc, "invalid VOPDY instruction");
5870 }
5871 return Error(ErrorLoc, "invalid operand for instruction");
5872 }
5873
5874 case Match_MnemonicFail:
5875 llvm_unreachable("Invalid instructions should have been handled already");
5876 }
5877 llvm_unreachable("Implement any new match types added!");
5878}
5879
5880bool AMDGPUAsmParser::ParseAsAbsoluteExpression(uint32_t &Ret) {
5881 int64_t Tmp = -1;
5882 if (!isToken(AsmToken::Integer) && !isToken(AsmToken::Identifier)) {
5883 return true;
5884 }
5885 if (getParser().parseAbsoluteExpression(Tmp)) {
5886 return true;
5887 }
5888 Ret = static_cast<uint32_t>(Tmp);
5889 return false;
5890}
5891
5892bool AMDGPUAsmParser::ParseDirectiveAMDGCNTarget() {
5893 if (!getSTI().getTargetTriple().isAMDGCN())
5894 return TokError("directive only supported for amdgcn architecture");
5895
5896 std::string TargetIDDirective;
5897 SMLoc TargetStart = getTok().getLoc();
5898 if (getParser().parseEscapedString(TargetIDDirective))
5899 return true;
5900
5901 std::optional<AMDGPU::TargetID> MaybeParsed =
5902 AMDGPU::TargetID::parseTargetIDString(TargetIDDirective);
5903 if (!MaybeParsed)
5904 return getParser().Error(TargetStart,
5905 "malformed target id '" + TargetIDDirective + "'");
5906
5907 const AMDGPU::TargetID &ParsedTargetID = *MaybeParsed;
5908 const Triple &TT = getSTI().getTargetTriple();
5909
5910 // The processor named in the target id must be covered by the triple's
5911 // subarch.
5912 if (!AMDGPU::isCPUValidForSubArch(TT.getSubArch(),
5913 ParsedTargetID.getGPUKind())) {
5914 return getParser().Error(
5915 TargetStart, "target id '" + TargetIDDirective +
5916 "' specifies a processor that is not valid for "
5917 "subarch '" +
5918 TT.getArchName() + "'");
5919 }
5920
5921 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
5922 getTargetStreamer().getTargetID();
5923
5924 Triple DirectiveTriple(ParsedTargetID.getTargetTripleString());
5925 const Triple &STITriple = getSTI().getTargetTriple();
5926 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
5927 return getParser().Error(
5928 TargetStart, ".amdgcn_target " + Twine(ParsedTargetID.toString()) +
5929 " is incompatible with " +
5930 Twine(CurrentTargetID->toString()));
5931 }
5932
5933 // Error if the ISA version doesn't match
5934 StringRef DirectiveProcessor =
5935 AMDGPU::getArchNameAMDGCN(ParsedTargetID.getGPUKind());
5936 AMDGPU::IsaVersion DirectiveISA = AMDGPU::getIsaVersion(DirectiveProcessor);
5937 if (DirectiveISA != ISA) {
5938 return getParser().Error(TargetStart,
5939 ".amdgcn_target directive processor " +
5940 Twine(DirectiveProcessor) +
5941 " does not match the specified processor " +
5942 Twine(getSTI().getCPU()));
5943 }
5944
5945 // Warn if sramecc or xnack mismatch. These do not change the encoding.
5947 ParsedTargetID.getXnackSetting(),
5948 CurrentTargetID->getXnackSetting())) {
5949 Warning(TargetStart,
5950 ".amdgcn_target directive has conflicting xnack settings");
5951 }
5953 ParsedTargetID.getSramEccSetting(),
5954 CurrentTargetID->getSramEccSetting())) {
5955 Warning(TargetStart,
5956 ".amdgcn_target directive has conflicting sramecc settings");
5957 }
5958
5959 // Update the target streamer's TargetID with settings from the directive.
5960 // We don't update the MCSubtargetInfo because we've already validated
5961 // that the directive matches the command-line CPU.
5962 getTargetStreamer().getTargetID()->setXnackSetting(
5963 ParsedTargetID.getXnackSetting());
5964 getTargetStreamer().getTargetID()->setSramEccSetting(
5965 ParsedTargetID.getSramEccSetting());
5966
5967 return false;
5968}
5969
5970bool AMDGPUAsmParser::OutOfRangeError(SMRange Range) {
5971 return Error(Range.Start, "value out of range", Range);
5972}
5973
5974bool AMDGPUAsmParser::calculateGPRBlocks(
5975 const FeatureBitset &Features, const MCExpr *VCCUsed,
5976 const MCExpr *FlatScrUsed, bool XNACKUsed,
5977 std::optional<bool> EnableWavefrontSize32, const MCExpr *NextFreeVGPR,
5978 SMRange VGPRRange, const MCExpr *NextFreeSGPR, SMRange SGPRRange,
5979 const MCExpr *&VGPRBlocks, const MCExpr *&SGPRBlocks) {
5980 // TODO(scott.linder): These calculations are duplicated from
5981 // AMDGPUAsmPrinter::getSIProgramInfo and could be unified.
5982 MCContext &Ctx = getContext();
5983
5984 const MCExpr *NumSGPRs = NextFreeSGPR;
5985 int64_t EvaluatedSGPRs;
5986
5987 if (ISA.Major >= 10)
5989 else {
5990 unsigned MaxAddressableNumSGPRs = AMDGPU::getAddressableNumSGPRs(Gfx);
5991
5992 if (NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) && ISA.Major >= 8 &&
5993 !Features.test(FeatureSGPRInitBug) &&
5994 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
5995 return OutOfRangeError(SGPRRange);
5996
5997 const MCExpr *ExtraSGPRs =
5998 AMDGPUMCExpr::createExtraSGPRs(VCCUsed, FlatScrUsed, XNACKUsed, Ctx);
5999 NumSGPRs = MCBinaryExpr::createAdd(NumSGPRs, ExtraSGPRs, Ctx);
6000
6001 if (NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
6002 (ISA.Major <= 7 || Features.test(FeatureSGPRInitBug)) &&
6003 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6004 return OutOfRangeError(SGPRRange);
6005
6006 if (Features.test(FeatureSGPRInitBug))
6007 NumSGPRs =
6009 }
6010
6011 // The MCExpr equivalent of getNumSGPRBlocks/getNumVGPRBlocks:
6012 // (alignTo(max(1u, NumGPR), GPREncodingGranule) / GPREncodingGranule) - 1
6013 auto GetNumGPRBlocks = [&Ctx](const MCExpr *NumGPR,
6014 unsigned Granule) -> const MCExpr * {
6015 const MCExpr *OneConst = MCConstantExpr::create(1ul, Ctx);
6016 const MCExpr *GranuleConst = MCConstantExpr::create(Granule, Ctx);
6017 const MCExpr *MaxNumGPR = AMDGPUMCExpr::createMax({NumGPR, OneConst}, Ctx);
6018 const MCExpr *AlignToGPR =
6019 AMDGPUMCExpr::createAlignTo(MaxNumGPR, GranuleConst, Ctx);
6020 const MCExpr *DivGPR =
6021 MCBinaryExpr::createDiv(AlignToGPR, GranuleConst, Ctx);
6022 const MCExpr *SubGPR = MCBinaryExpr::createSub(DivGPR, OneConst, Ctx);
6023 return SubGPR;
6024 };
6025
6026 VGPRBlocks = GetNumGPRBlocks(
6027 NextFreeVGPR,
6028 IsaInfo::getVGPREncodingGranule(getSTI(), EnableWavefrontSize32));
6029 SGPRBlocks =
6030 GetNumGPRBlocks(NumSGPRs, IsaInfo::getSGPREncodingGranule(getSTI()));
6031
6032 return false;
6033}
6034
6035bool AMDGPUAsmParser::ParseDirectiveAMDHSAKernel() {
6036 if (!getSTI().getTargetTriple().isAMDGCN())
6037 return TokError("directive only supported for amdgcn architecture");
6038
6039 if (!isHsaAbi(getSTI()))
6040 return TokError("directive only supported for amdhsa OS");
6041
6042 StringRef KernelName;
6043 if (getParser().parseIdentifier(KernelName))
6044 return true;
6045
6046 // Remember the kernel name so its prologue can be checked at end of file.
6047 // The matching label may have been parsed already or may follow later.
6048 AMDHSAKernelSymbols.insert(getContext().getOrCreateSymbol(KernelName));
6049
6050 AMDGPU::MCKernelDescriptor KD =
6052 &getSTI(), getContext());
6053
6054 StringSet<> Seen;
6055
6056 const MCExpr *ZeroExpr = MCConstantExpr::create(0, getContext());
6057 const MCExpr *OneExpr = MCConstantExpr::create(1, getContext());
6058
6059 SMRange VGPRRange;
6060 const MCExpr *NextFreeVGPR = ZeroExpr;
6061 const MCExpr *AccumOffset = MCConstantExpr::create(0, getContext());
6062 const MCExpr *NamedBarCnt = ZeroExpr;
6063 uint64_t SharedVGPRCount = 0;
6064 uint64_t PreloadLength = 0;
6065 uint64_t PreloadOffset = 0;
6066 SMRange SGPRRange;
6067 const MCExpr *NextFreeSGPR = ZeroExpr;
6068
6069 // Count the number of user SGPRs implied from the enabled feature bits.
6070 unsigned ImpliedUserSGPRCount = 0;
6071
6072 // Track if the asm explicitly contains the directive for the user SGPR
6073 // count.
6074 std::optional<unsigned> ExplicitUserSGPRCount;
6075 const MCExpr *ReserveVCC = OneExpr;
6076 const MCExpr *ReserveFlatScr = OneExpr;
6077 std::optional<bool> EnableWavefrontSize32;
6078
6079 while (true) {
6080 while (trySkipToken(AsmToken::EndOfStatement))
6081 ;
6082
6083 StringRef ID;
6084 SMRange IDRange = getTok().getLocRange();
6085 if (!parseId(ID, "expected .amdhsa_ directive or .end_amdhsa_kernel"))
6086 return true;
6087
6088 if (ID == ".end_amdhsa_kernel")
6089 break;
6090
6091 if (!Seen.insert(ID).second)
6092 return TokError(".amdhsa_ directives cannot be repeated");
6093
6094 SMLoc ValStart = getLoc();
6095 const MCExpr *ExprVal;
6096 if (getParser().parseExpression(ExprVal))
6097 return true;
6098 SMLoc ValEnd = getLoc();
6099 SMRange ValRange = SMRange(ValStart, ValEnd);
6100
6101 int64_t IVal = 0;
6102 uint64_t Val = IVal;
6103 bool EvaluatableExpr;
6104 if ((EvaluatableExpr = ExprVal->evaluateAsAbsolute(IVal))) {
6105 if (IVal < 0)
6106 return OutOfRangeError(ValRange);
6107 Val = IVal;
6108 }
6109
6110#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE) \
6111 if (!isUInt<ENTRY##_WIDTH>(Val)) \
6112 return OutOfRangeError(RANGE); \
6113 AMDGPU::MCKernelDescriptor::bits_set(FIELD, VALUE, ENTRY##_SHIFT, ENTRY, \
6114 getContext());
6115
6116// Some fields use the parsed value immediately which requires the expression to
6117// be solvable.
6118#define EXPR_RESOLVE_OR_ERROR(RESOLVED) \
6119 if (!(RESOLVED)) \
6120 return Error(IDRange.Start, "directive should have resolvable expression", \
6121 IDRange);
6122
6123 if (ID == ".amdhsa_group_segment_fixed_size") {
6125 CHAR_BIT>(Val))
6126 return OutOfRangeError(ValRange);
6127 KD.group_segment_fixed_size = ExprVal;
6128 } else if (ID == ".amdhsa_private_segment_fixed_size") {
6130 CHAR_BIT>(Val))
6131 return OutOfRangeError(ValRange);
6132 KD.private_segment_fixed_size = ExprVal;
6133 } else if (ID == ".amdhsa_kernarg_size") {
6134 if (!isUInt<sizeof(kernel_descriptor_t::kernarg_size) * CHAR_BIT>(Val))
6135 return OutOfRangeError(ValRange);
6136 KD.kernarg_size = ExprVal;
6137 } else if (ID == ".amdhsa_user_sgpr_count") {
6138 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6139 ExplicitUserSGPRCount = Val;
6140 } else if (ID == ".amdhsa_user_sgpr_private_segment_buffer") {
6141 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6143 return Error(IDRange.Start,
6144 "directive is not supported with architected flat scratch",
6145 IDRange);
6147 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER,
6148 ExprVal, ValRange);
6149 if (Val)
6150 ImpliedUserSGPRCount += 4;
6151 } else if (ID == ".amdhsa_user_sgpr_kernarg_preload_length") {
6152 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6153 if (!hasKernargPreload())
6154 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6155
6156 if (Val > getMaxNumUserSGPRs())
6157 return OutOfRangeError(ValRange);
6158 PARSE_BITS_ENTRY(KD.kernarg_preload, KERNARG_PRELOAD_SPEC_LENGTH, ExprVal,
6159 ValRange);
6160 if (Val) {
6161 ImpliedUserSGPRCount += Val;
6162 PreloadLength = Val;
6163 }
6164 } else if (ID == ".amdhsa_user_sgpr_kernarg_preload_offset") {
6165 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6166 if (!hasKernargPreload())
6167 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6168
6169 if (Val >= 1024)
6170 return OutOfRangeError(ValRange);
6171 PARSE_BITS_ENTRY(KD.kernarg_preload, KERNARG_PRELOAD_SPEC_OFFSET, ExprVal,
6172 ValRange);
6173 if (Val)
6174 PreloadOffset = Val;
6175 } else if (ID == ".amdhsa_user_sgpr_dispatch_ptr") {
6176 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6178 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR, ExprVal,
6179 ValRange);
6180 if (Val)
6181 ImpliedUserSGPRCount += 2;
6182 } else if (ID == ".amdhsa_user_sgpr_queue_ptr") {
6183 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6185 KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR, ExprVal,
6186 ValRange);
6187 if (Val)
6188 ImpliedUserSGPRCount += 2;
6189 } else if (ID == ".amdhsa_user_sgpr_kernarg_segment_ptr") {
6190 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6192 KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR,
6193 ExprVal, ValRange);
6194 if (Val)
6195 ImpliedUserSGPRCount += 2;
6196 } else if (ID == ".amdhsa_user_sgpr_dispatch_id") {
6197 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6199 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID, ExprVal,
6200 ValRange);
6201 if (Val)
6202 ImpliedUserSGPRCount += 2;
6203 } else if (ID == ".amdhsa_user_sgpr_flat_scratch_init") {
6205 return Error(IDRange.Start,
6206 "directive is not supported with architected flat scratch",
6207 IDRange);
6208 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6210 KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT,
6211 ExprVal, ValRange);
6212 if (Val)
6213 ImpliedUserSGPRCount += 2;
6214 } else if (ID == ".amdhsa_user_sgpr_private_segment_size") {
6215 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6217 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE,
6218 ExprVal, ValRange);
6219 if (Val)
6220 ImpliedUserSGPRCount += 1;
6221 } else if (ID == ".amdhsa_wavefront_size32") {
6222 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6223 if (ISA.Major < 10)
6224 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6225 EnableWavefrontSize32 = Val;
6227 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32, ExprVal,
6228 ValRange);
6229 } else if (ID == ".amdhsa_uses_dynamic_stack") {
6231 KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK, ExprVal,
6232 ValRange);
6233 } else if (ID == ".amdhsa_system_sgpr_private_segment_wavefront_offset") {
6235 return Error(IDRange.Start,
6236 "directive is not supported with architected flat scratch",
6237 IDRange);
6239 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6240 ValRange);
6241 } else if (ID == ".amdhsa_enable_private_segment") {
6243 return Error(
6244 IDRange.Start,
6245 "directive is not supported without architected flat scratch",
6246 IDRange);
6248 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6249 ValRange);
6250 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_x") {
6252 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_X, ExprVal,
6253 ValRange);
6254 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_y") {
6256 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Y, ExprVal,
6257 ValRange);
6258 } else if (ID == ".amdhsa_system_sgpr_workgroup_id_z") {
6260 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Z, ExprVal,
6261 ValRange);
6262 } else if (ID == ".amdhsa_system_sgpr_workgroup_info") {
6264 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_INFO, ExprVal,
6265 ValRange);
6266 } else if (ID == ".amdhsa_system_vgpr_workitem_id") {
6268 COMPUTE_PGM_RSRC2_ENABLE_VGPR_WORKITEM_ID, ExprVal,
6269 ValRange);
6270 } else if (ID == ".amdhsa_next_free_vgpr") {
6271 VGPRRange = ValRange;
6272 NextFreeVGPR = ExprVal;
6273 } else if (ID == ".amdhsa_next_free_sgpr") {
6274 SGPRRange = ValRange;
6275 NextFreeSGPR = ExprVal;
6276 } else if (ID == ".amdhsa_accum_offset") {
6277 if (!isGFX90A())
6278 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6279 AccumOffset = ExprVal;
6280 } else if (ID == ".amdhsa_named_barrier_count") {
6281 if (!isGFX1250Plus())
6282 return Error(IDRange.Start, "directive requires gfx1250+", IDRange);
6283 NamedBarCnt = ExprVal;
6284 } else if (ID == ".amdhsa_reserve_vcc") {
6285 if (EvaluatableExpr && !isUInt<1>(Val))
6286 return OutOfRangeError(ValRange);
6287 ReserveVCC = ExprVal;
6288 } else if (ID == ".amdhsa_reserve_flat_scratch") {
6289 if (ISA.Major < 7)
6290 return Error(IDRange.Start, "directive requires gfx7+", IDRange);
6292 return Error(IDRange.Start,
6293 "directive is not supported with architected flat scratch",
6294 IDRange);
6295 if (EvaluatableExpr && !isUInt<1>(Val))
6296 return OutOfRangeError(ValRange);
6297 ReserveFlatScr = ExprVal;
6298 } else if (ID == ".amdhsa_reserve_xnack_mask") {
6299 if (ISA.Major < 8)
6300 return Error(IDRange.Start, "directive requires gfx8+", IDRange);
6301 if (!isUInt<1>(Val))
6302 return OutOfRangeError(ValRange);
6303 bool XnackOn = getTargetStreamer().getTargetID()->isXnackOnOrAny();
6304 if (Val != XnackOn) {
6305 return getParser().Error(
6306 IDRange.Start,
6307 ".amdhsa_reserve_xnack_mask does not match target id", IDRange);
6308 }
6309 } else if (ID == ".amdhsa_float_round_mode_32") {
6311 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_32, ExprVal,
6312 ValRange);
6313 } else if (ID == ".amdhsa_float_round_mode_16_64") {
6315 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_16_64, ExprVal,
6316 ValRange);
6317 } else if (ID == ".amdhsa_float_denorm_mode_32") {
6319 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_32, ExprVal,
6320 ValRange);
6321 } else if (ID == ".amdhsa_float_denorm_mode_16_64") {
6323 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_16_64, ExprVal,
6324 ValRange);
6325 } else if (ID == ".amdhsa_dx10_clamp") {
6326 if (!getSTI().hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6327 return Error(IDRange.Start, "directive unsupported on gfx1170+",
6328 IDRange);
6330 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_DX10_CLAMP, ExprVal,
6331 ValRange);
6332 } else if (ID == ".amdhsa_ieee_mode") {
6333 if (!getSTI().hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6334 return Error(IDRange.Start, "directive unsupported on gfx1170+",
6335 IDRange);
6337 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_IEEE_MODE, ExprVal,
6338 ValRange);
6339 } else if (ID == ".amdhsa_fp16_overflow") {
6340 if (ISA.Major < 9)
6341 return Error(IDRange.Start, "directive requires gfx9+", IDRange);
6343 COMPUTE_PGM_RSRC1_GFX9_PLUS_FP16_OVFL, ExprVal,
6344 ValRange);
6345 } else if (ID == ".amdhsa_tg_split") {
6346 if (!isGFX90A())
6347 return Error(IDRange.Start, "directive requires gfx90a+", IDRange);
6348 PARSE_BITS_ENTRY(KD.compute_pgm_rsrc3, COMPUTE_PGM_RSRC3_GFX90A_TG_SPLIT,
6349 ExprVal, ValRange);
6350 } else if (ID == ".amdhsa_workgroup_processor_mode") {
6351 if (!supportsWGP(getSTI()))
6352 return Error(IDRange.Start,
6353 "directive unsupported on " + getSTI().getCPU(), IDRange);
6355 COMPUTE_PGM_RSRC1_GFX10_PLUS_WGP_MODE, ExprVal,
6356 ValRange);
6357 } else if (ID == ".amdhsa_memory_ordered") {
6358 if (ISA.Major < 10)
6359 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6361 COMPUTE_PGM_RSRC1_GFX10_PLUS_MEM_ORDERED, ExprVal,
6362 ValRange);
6363 } else if (ID == ".amdhsa_forward_progress") {
6364 if (ISA.Major < 10)
6365 return Error(IDRange.Start, "directive requires gfx10+", IDRange);
6367 COMPUTE_PGM_RSRC1_GFX10_PLUS_FWD_PROGRESS, ExprVal,
6368 ValRange);
6369 } else if (ID == ".amdhsa_shared_vgpr_count") {
6370 EXPR_RESOLVE_OR_ERROR(EvaluatableExpr);
6371 if (ISA.Major < 10 || ISA.Major >= 12)
6372 return Error(IDRange.Start, "directive requires gfx10 or gfx11",
6373 IDRange);
6374 SharedVGPRCount = Val;
6376 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT, ExprVal,
6377 ValRange);
6378 } else if (ID == ".amdhsa_inst_pref_size") {
6379 if (ISA.Major < 11)
6380 return Error(IDRange.Start, "directive requires gfx11+", IDRange);
6381 if (ISA.Major == 11) {
6383 COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE, ExprVal,
6384 ValRange);
6385 } else {
6387 COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE, ExprVal,
6388 ValRange);
6389 }
6390 } else if (ID == ".amdhsa_exception_fp_ieee_invalid_op") {
6393 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INVALID_OPERATION,
6394 ExprVal, ValRange);
6395 } else if (ID == ".amdhsa_exception_fp_denorm_src") {
6397 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_FP_DENORMAL_SOURCE,
6398 ExprVal, ValRange);
6399 } else if (ID == ".amdhsa_exception_fp_ieee_div_zero") {
6402 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_DIVISION_BY_ZERO,
6403 ExprVal, ValRange);
6404 } else if (ID == ".amdhsa_exception_fp_ieee_overflow") {
6406 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_OVERFLOW,
6407 ExprVal, ValRange);
6408 } else if (ID == ".amdhsa_exception_fp_ieee_underflow") {
6410 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_UNDERFLOW,
6411 ExprVal, ValRange);
6412 } else if (ID == ".amdhsa_exception_fp_ieee_inexact") {
6414 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INEXACT,
6415 ExprVal, ValRange);
6416 } else if (ID == ".amdhsa_exception_int_div_zero") {
6418 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_INT_DIVIDE_BY_ZERO,
6419 ExprVal, ValRange);
6420 } else if (ID == ".amdhsa_round_robin_scheduling") {
6421 if (ISA.Major < 12)
6422 return Error(IDRange.Start, "directive requires gfx12+", IDRange);
6424 COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN, ExprVal,
6425 ValRange);
6426 } else {
6427 return Error(IDRange.Start, "unknown .amdhsa_kernel directive", IDRange);
6428 }
6429
6430#undef PARSE_BITS_ENTRY
6431 }
6432
6433 if (!Seen.contains(".amdhsa_next_free_vgpr"))
6434 return TokError(".amdhsa_next_free_vgpr directive is required");
6435
6436 if (!Seen.contains(".amdhsa_next_free_sgpr"))
6437 return TokError(".amdhsa_next_free_sgpr directive is required");
6438
6439 unsigned UserSGPRCount = ExplicitUserSGPRCount.value_or(ImpliedUserSGPRCount);
6440 if (UserSGPRCount > getMaxNumUserSGPRs())
6441 return TokError("too many user SGPRs enabled, found " +
6442 Twine(UserSGPRCount) + ", but only " +
6443 Twine(getMaxNumUserSGPRs()) + " are supported.");
6444
6445 // Consider the case where the total number of UserSGPRs with trailing
6446 // allocated preload SGPRs, is greater than the number of explicitly
6447 // referenced SGPRs.
6448 if (PreloadLength) {
6449 MCContext &Ctx = getContext();
6450 NextFreeSGPR = AMDGPUMCExpr::createMax(
6451 {NextFreeSGPR, MCConstantExpr::create(UserSGPRCount, Ctx)}, Ctx);
6452 }
6453
6454 const MCExpr *VGPRBlocks;
6455 const MCExpr *SGPRBlocks;
6456 if (calculateGPRBlocks(getFeatureBits(), ReserveVCC, ReserveFlatScr,
6457 getTargetStreamer().getTargetID()->isXnackOnOrAny(),
6458 EnableWavefrontSize32, NextFreeVGPR, VGPRRange,
6459 NextFreeSGPR, SGPRRange, VGPRBlocks, SGPRBlocks))
6460 return true;
6461
6462 int64_t EvaluatedVGPRBlocks;
6463 bool VGPRBlocksEvaluatable =
6464 VGPRBlocks->evaluateAsAbsolute(EvaluatedVGPRBlocks);
6465 if (VGPRBlocksEvaluatable &&
6467 static_cast<uint64_t>(EvaluatedVGPRBlocks))) {
6468 return OutOfRangeError(VGPRRange);
6469 }
6471 KD.compute_pgm_rsrc1, VGPRBlocks,
6472 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT_SHIFT,
6473 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT, getContext());
6474
6475 int64_t EvaluatedSGPRBlocks;
6476 if (SGPRBlocks->evaluateAsAbsolute(EvaluatedSGPRBlocks) &&
6478 static_cast<uint64_t>(EvaluatedSGPRBlocks)))
6479 return OutOfRangeError(SGPRRange);
6481 KD.compute_pgm_rsrc1, SGPRBlocks,
6482 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT_SHIFT,
6483 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT, getContext());
6484
6485 if (ExplicitUserSGPRCount && ImpliedUserSGPRCount > *ExplicitUserSGPRCount)
6486 return TokError("amdgpu_user_sgpr_count smaller than implied by "
6487 "enabled user SGPRs");
6488
6489 if (isGFX1250Plus()) {
6492 MCConstantExpr::create(UserSGPRCount, getContext()),
6493 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT_SHIFT,
6494 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT, getContext());
6495 } else {
6498 MCConstantExpr::create(UserSGPRCount, getContext()),
6499 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT_SHIFT,
6500 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT, getContext());
6501 }
6502
6503 int64_t IVal = 0;
6504 if (!KD.kernarg_size->evaluateAsAbsolute(IVal))
6505 return TokError("Kernarg size should be resolvable");
6506 uint64_t kernarg_size = IVal;
6507 if (PreloadLength && kernarg_size &&
6508 (PreloadLength * 4 + PreloadOffset * 4 > kernarg_size))
6509 return TokError("Kernarg preload length + offset is larger than the "
6510 "kernarg segment size");
6511
6512 if (isGFX90A()) {
6513 if (!Seen.contains(".amdhsa_accum_offset"))
6514 return TokError(".amdhsa_accum_offset directive is required");
6515 int64_t EvaluatedAccum;
6516 bool AccumEvaluatable = AccumOffset->evaluateAsAbsolute(EvaluatedAccum);
6517 uint64_t UEvaluatedAccum = EvaluatedAccum;
6518 if (AccumEvaluatable &&
6519 (UEvaluatedAccum < 4 || UEvaluatedAccum > 256 || (UEvaluatedAccum & 3)))
6520 return TokError("accum_offset should be in range [4..256] in "
6521 "increments of 4");
6522
6523 int64_t EvaluatedNumVGPR;
6524 if (NextFreeVGPR->evaluateAsAbsolute(EvaluatedNumVGPR) &&
6525 AccumEvaluatable &&
6526 UEvaluatedAccum >
6527 alignTo(std::max((uint64_t)1, (uint64_t)EvaluatedNumVGPR), 4))
6528 return TokError("accum_offset exceeds total VGPR allocation");
6529 const MCExpr *AdjustedAccum = MCBinaryExpr::createSub(
6531 AccumOffset, MCConstantExpr::create(4, getContext()), getContext()),
6534 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT,
6535 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
6536 getContext());
6537 }
6538
6539 if (isGFX1250Plus())
6541 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT,
6542 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
6543 getContext());
6544
6545 if (ISA.Major >= 10 && ISA.Major < 12) {
6546 // SharedVGPRCount < 16 checked by PARSE_ENTRY_BITS
6547 if (SharedVGPRCount && EnableWavefrontSize32 && *EnableWavefrontSize32) {
6548 return TokError("shared_vgpr_count directive not valid on "
6549 "wavefront size 32");
6550 }
6551
6552 if (VGPRBlocksEvaluatable &&
6553 (SharedVGPRCount * 2 + static_cast<uint64_t>(EvaluatedVGPRBlocks) >
6554 63)) {
6555 return TokError("shared_vgpr_count*2 + "
6556 "compute_pgm_rsrc1.GRANULATED_WORKITEM_VGPR_COUNT cannot "
6557 "exceed 63\n");
6558 }
6559 }
6560
6561 emitTargetDirective();
6562 getTargetStreamer().EmitAmdhsaKernelDescriptor(getSTI(), KernelName, KD,
6563 NextFreeVGPR, NextFreeSGPR,
6564 ReserveVCC, ReserveFlatScr);
6565 return false;
6566}
6567
6568bool AMDGPUAsmParser::ParseDirectiveAMDHSACodeObjectVersion() {
6569 uint32_t Version;
6570 if (ParseAsAbsoluteExpression(Version))
6571 return true;
6572
6573 getTargetStreamer().EmitDirectiveAMDHSACodeObjectVersion(Version);
6574 emitTargetDirective();
6575 return false;
6576}
6577
6578bool AMDGPUAsmParser::ParseAMDKernelCodeTValue(StringRef ID,
6579 AMDGPUMCKernelCodeT &C) {
6580 // max_scratch_backing_memory_byte_size is deprecated. Ignore it while parsing
6581 // assembly for backwards compatibility.
6582 if (ID == "max_scratch_backing_memory_byte_size") {
6583 Parser.eatToEndOfStatement();
6584 return false;
6585 }
6586
6587 SmallString<40> ErrStr;
6588 raw_svector_ostream Err(ErrStr);
6589 if (!C.ParseKernelCodeT(ID, getParser(), Err)) {
6590 return TokError(Err.str());
6591 }
6592 Lex();
6593
6594 if (ID == "enable_wavefront_size32") {
6595 if (C.code_properties & AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32) {
6596 if (!isGFX10Plus())
6597 return TokError("enable_wavefront_size32=1 is only allowed on GFX10+");
6598 if (!isWave32())
6599 return TokError("enable_wavefront_size32=1 requires +WavefrontSize32");
6600 } else {
6601 if (!isWave64())
6602 return TokError("enable_wavefront_size32=0 requires +WavefrontSize64");
6603 }
6604 }
6605
6606 if (ID == "wavefront_size") {
6607 if (C.wavefront_size == 5) {
6608 if (!isGFX10Plus())
6609 return TokError("wavefront_size=5 is only allowed on GFX10+");
6610 if (!isWave32())
6611 return TokError("wavefront_size=5 requires +WavefrontSize32");
6612 } else if (C.wavefront_size == 6) {
6613 if (!isWave64())
6614 return TokError("wavefront_size=6 requires +WavefrontSize64");
6615 }
6616 }
6617
6618 return false;
6619}
6620
6621bool AMDGPUAsmParser::ParseDirectiveAMDKernelCodeT() {
6622 AMDGPUMCKernelCodeT KernelCode;
6623 KernelCode.initDefault(getSTI(), getContext());
6624
6625 while (true) {
6626 // Lex EndOfStatement. This is in a while loop, because lexing a comment
6627 // will set the current token to EndOfStatement.
6628 while (trySkipToken(AsmToken::EndOfStatement))
6629 ;
6630
6631 StringRef ID;
6632 if (!parseId(ID, "expected value identifier or .end_amd_kernel_code_t"))
6633 return true;
6634
6635 if (ID == ".end_amd_kernel_code_t")
6636 break;
6637
6638 if (ParseAMDKernelCodeTValue(ID, KernelCode))
6639 return true;
6640 }
6641
6642 KernelCode.validate(&getSTI(), getContext());
6643 getTargetStreamer().EmitAMDKernelCodeT(KernelCode);
6644
6645 return false;
6646}
6647
6648bool AMDGPUAsmParser::ParseDirectiveAMDGPUHsaKernel() {
6649 StringRef KernelName;
6650 if (!parseId(KernelName, "expected symbol name"))
6651 return true;
6652
6653 getTargetStreamer().EmitAMDGPUSymbolType(KernelName,
6655
6656 KernelScope.initialize(getContext());
6657 return false;
6658}
6659
6660bool AMDGPUAsmParser::ParseDirectiveISAVersion() {
6661 if (!getSTI().getTargetTriple().isAMDGCN()) {
6662 return Error(getLoc(),
6663 ".amd_amdgpu_isa directive is not available on non-amdgcn "
6664 "architectures");
6665 }
6666
6667 StringRef TargetIDDirective = getLexer().getTok().getStringContents();
6668
6669 std::optional<AMDGPU::TargetID> MaybeParsed =
6670 AMDGPU::TargetID::parseTargetIDString(TargetIDDirective);
6671 if (!MaybeParsed)
6672 return Error(getParser().getTok().getLoc(),
6673 "malformed target id '" + TargetIDDirective + "'");
6674
6675 const AMDGPU::TargetID &ParsedTargetID = *MaybeParsed;
6676 const Triple &TT = getSTI().getTargetTriple();
6677
6678 // The processor named in the target id must be covered by the triple's
6679 // subarch.
6680 if (!AMDGPU::isCPUValidForSubArch(TT.getSubArch(),
6681 ParsedTargetID.getGPUKind())) {
6682 return Error(getParser().getTok().getLoc(),
6683 "target id '" + TargetIDDirective +
6684 "' specifies a processor that is not valid for subarch '" +
6685 TT.getArchName() + "'");
6686 }
6687
6688 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6689 getTargetStreamer().getTargetID();
6690
6691 Triple DirectiveTriple(ParsedTargetID.getTargetTripleString());
6692 const Triple &STITriple = getSTI().getTargetTriple();
6693 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6694 return Error(getParser().getTok().getLoc(),
6695 ".amd_amdgpu_isa " + Twine(ParsedTargetID.toString()) +
6696 " is incompatible with " +
6697 Twine(CurrentTargetID->toString()));
6698 }
6699
6700 // Error if the ISA version doesn't match
6701 StringRef DirectiveProcessor =
6702 AMDGPU::getArchNameAMDGCN(ParsedTargetID.getGPUKind());
6703 AMDGPU::IsaVersion DirectiveISA = AMDGPU::getIsaVersion(DirectiveProcessor);
6704 if (DirectiveISA != ISA) {
6705 return Error(getParser().getTok().getLoc(),
6706 ".amd_amdgpu_isa directive processor " +
6707 Twine(DirectiveProcessor) +
6708 " does not match the specified processor " +
6709 Twine(getSTI().getCPU()));
6710 }
6711
6712 getTargetStreamer().EmitISAVersion();
6713 Lex();
6714
6715 return false;
6716}
6717
6718bool AMDGPUAsmParser::ParseDirectiveHSAMetadata() {
6719 assert(isHsaAbi(getSTI()));
6720
6721 std::string HSAMetadataString;
6722 if (ParseToEndDirective(HSAMD::V3::AssemblerDirectiveBegin,
6723 HSAMD::V3::AssemblerDirectiveEnd, HSAMetadataString))
6724 return true;
6725
6726 if (!getTargetStreamer().EmitHSAMetadataV3(HSAMetadataString))
6727 return Error(getLoc(), "invalid HSA metadata");
6728
6729 return false;
6730}
6731
6732/// Common code to parse out a block of text (typically YAML) between start and
6733/// end directives.
6734bool AMDGPUAsmParser::ParseToEndDirective(const char *AssemblerDirectiveBegin,
6735 const char *AssemblerDirectiveEnd,
6736 std::string &CollectString) {
6737
6738 raw_string_ostream CollectStream(CollectString);
6739
6740 getLexer().setSkipSpace(false);
6741
6742 bool FoundEnd = false;
6743 while (!isToken(AsmToken::Eof)) {
6744 while (isToken(AsmToken::Space)) {
6745 CollectStream << getTokenStr();
6746 Lex();
6747 }
6748
6749 if (trySkipId(AssemblerDirectiveEnd)) {
6750 FoundEnd = true;
6751 break;
6752 }
6753
6754 CollectStream << Parser.parseStringToEndOfStatement()
6755 << getContext().getAsmInfo().getSeparatorString();
6756
6757 Parser.eatToEndOfStatement();
6758 }
6759
6760 getLexer().setSkipSpace(true);
6761
6762 if (isToken(AsmToken::Eof) && !FoundEnd) {
6763 return TokError(Twine("expected directive ") +
6764 Twine(AssemblerDirectiveEnd) + Twine(" not found"));
6765 }
6766
6767 return false;
6768}
6769
6770/// Parse the assembler directive for new MsgPack-format PAL metadata.
6771bool AMDGPUAsmParser::ParseDirectivePALMetadataBegin() {
6772 std::string String;
6773 if (ParseToEndDirective(AMDGPU::PALMD::AssemblerDirectiveBegin,
6775 return true;
6776
6777 auto *PALMetadata = getTargetStreamer().getPALMetadata();
6778 if (!PALMetadata->setFromString(String))
6779 return Error(getLoc(), "invalid PAL metadata");
6780 return false;
6781}
6782
6783/// Parse the assembler directive for old linear-format PAL metadata.
6784bool AMDGPUAsmParser::ParseDirectivePALMetadata() {
6785 if (getSTI().getTargetTriple().getOS() != Triple::AMDPAL) {
6786 return Error(getLoc(), (Twine(PALMD::AssemblerDirective) +
6787 Twine(" directive is "
6788 "not available on non-amdpal OSes"))
6789 .str());
6790 }
6791
6792 auto *PALMetadata = getTargetStreamer().getPALMetadata();
6793 PALMetadata->setLegacy();
6794 for (;;) {
6795 uint32_t Key, Value;
6796 if (ParseAsAbsoluteExpression(Key)) {
6797 return TokError(Twine("invalid value in ") +
6799 }
6800 if (!trySkipToken(AsmToken::Comma)) {
6801 return TokError(Twine("expected an even number of values in ") +
6803 }
6804 if (ParseAsAbsoluteExpression(Value)) {
6805 return TokError(Twine("invalid value in ") +
6807 }
6808 PALMetadata->setRegister(Key, Value);
6809 if (!trySkipToken(AsmToken::Comma))
6810 break;
6811 }
6812 return false;
6813}
6814
6815/// ParseDirectiveAMDGPULDS
6816/// ::= .amdgpu_lds identifier ',' size_expression [',' align_expression]
6817bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
6818 if (getParser().checkForValidSection())
6819 return true;
6820
6821 StringRef Name;
6822 SMLoc NameLoc = getLoc();
6823 if (getParser().parseIdentifier(Name))
6824 return TokError("expected identifier in directive");
6825
6826 MCSymbol *Symbol = getContext().getOrCreateSymbol(Name);
6827 if (getParser().parseComma())
6828 return true;
6829
6830 unsigned LocalMemorySize = AMDGPU::IsaInfo::getLocalMemorySize(getSTI());
6831
6832 int64_t Size;
6833 SMLoc SizeLoc = getLoc();
6834 if (getParser().parseAbsoluteExpression(Size))
6835 return true;
6836 if (Size < 0)
6837 return Error(SizeLoc, "size must be non-negative");
6838 if (Size > LocalMemorySize)
6839 return Error(SizeLoc, "size is too large");
6840
6841 int64_t Alignment = 4;
6842 if (trySkipToken(AsmToken::Comma)) {
6843 SMLoc AlignLoc = getLoc();
6844 if (getParser().parseAbsoluteExpression(Alignment))
6845 return true;
6846 if (Alignment < 0 || !isPowerOf2_64(Alignment))
6847 return Error(AlignLoc, "alignment must be a power of two");
6848
6849 // Alignment larger than the size of LDS is possible in theory, as long
6850 // as the linker manages to place to symbol at address 0, but we do want
6851 // to make sure the alignment fits nicely into a 32-bit integer.
6852 if (Alignment >= 1u << 31)
6853 return Error(AlignLoc, "alignment is too large");
6854 }
6855
6856 if (parseEOL())
6857 return true;
6858
6859 Symbol->redefineIfPossible();
6860 if (!Symbol->isUndefined())
6861 return Error(NameLoc, "invalid symbol redefinition");
6862
6863 getTargetStreamer().emitAMDGPULDS(Symbol, Size, Align(Alignment));
6864 return false;
6865}
6866
6867bool AMDGPUAsmParser::ParseDirectiveAMDGPUInfo() {
6868 if (getParser().checkForValidSection())
6869 return true;
6870
6871 StringRef FuncName;
6872 if (getParser().parseIdentifier(FuncName))
6873 return TokError("expected symbol name after .amdgpu_info");
6874
6875 MCSymbol *FuncSym = getContext().getOrCreateSymbol(FuncName);
6876 AMDGPU::InfoSectionData ParsedInfoData;
6877 AMDGPU::FuncInfo FI;
6878 FI.Sym = FuncSym;
6879 bool HasScalarAttrs = false;
6880
6881 while (true) {
6882 while (trySkipToken(AsmToken::EndOfStatement))
6883 ;
6884
6885 StringRef ID;
6886 SMLoc IDLoc = getLoc();
6887 if (!parseId(ID, "expected directive or .end_amdgpu_info"))
6888 return true;
6889
6890 if (ID == ".end_amdgpu_info")
6891 break;
6892
6893 // Every per-entry directive shares the `.amdgpu_` namespace prefix; strip
6894 // it once and dispatch on the distinguishing suffix below. The unstripped
6895 // ID is preserved for diagnostics.
6896 StringRef Dir = ID;
6897 if (!Dir.consume_front(".amdgpu_"))
6898 return Error(IDLoc, "unknown .amdgpu_info directive '" + ID + "'");
6899
6900 if (Dir == "flags") {
6901 int64_t Val;
6902 if (getParser().parseAbsoluteExpression(Val))
6903 return true;
6904 auto Flags = static_cast<AMDGPU::FuncInfoFlags>(Val);
6905 FI.UsesVCC = !!(Flags & AMDGPU::FuncInfoFlags::FUNC_USES_VCC);
6906 FI.UsesFlatScratch =
6907 !!(Flags & AMDGPU::FuncInfoFlags::FUNC_USES_FLAT_SCRATCH);
6908 FI.HasDynStack = !!(Flags & AMDGPU::FuncInfoFlags::FUNC_HAS_DYN_STACK);
6909 HasScalarAttrs = true;
6910 } else if (Dir == "num_sgpr") {
6911 int64_t Val;
6912 if (getParser().parseAbsoluteExpression(Val))
6913 return true;
6914 FI.NumSGPR = static_cast<uint32_t>(Val);
6915 HasScalarAttrs = true;
6916 } else if (Dir == "num_vgpr") {
6917 int64_t Val;
6918 if (getParser().parseAbsoluteExpression(Val))
6919 return true;
6920 FI.NumArchVGPR = static_cast<uint32_t>(Val);
6921 HasScalarAttrs = true;
6922 } else if (Dir == "num_agpr") {
6923 int64_t Val;
6924 if (getParser().parseAbsoluteExpression(Val))
6925 return true;
6926 FI.NumAccVGPR = static_cast<uint32_t>(Val);
6927 HasScalarAttrs = true;
6928 } else if (Dir == "private_segment_size") {
6929 int64_t Val;
6930 if (getParser().parseAbsoluteExpression(Val))
6931 return true;
6932 FI.PrivateSegmentSize = static_cast<uint32_t>(Val);
6933 HasScalarAttrs = true;
6934 } else if (Dir == "use") {
6935 StringRef ResName;
6936 if (getParser().parseIdentifier(ResName))
6937 return TokError("expected resource symbol for .amdgpu_use");
6938 ParsedInfoData.Uses.push_back(
6939 {FuncSym, getContext().getOrCreateSymbol(ResName)});
6940 } else if (Dir == "call") {
6941 StringRef DstName;
6942 if (getParser().parseIdentifier(DstName))
6943 return TokError("expected callee symbol for .amdgpu_call");
6944 ParsedInfoData.Calls.push_back(
6945 {FuncSym, getContext().getOrCreateSymbol(DstName)});
6946 } else if (Dir == "indirect_call") {
6947 std::string TypeId;
6948 if (getParser().parseEscapedString(TypeId))
6949 return TokError("expected type ID string for .amdgpu_indirect_call");
6950 ParsedInfoData.IndirectCalls.push_back({FuncSym, std::move(TypeId)});
6951 } else if (Dir == "typeid") {
6952 std::string TypeId;
6953 if (getParser().parseEscapedString(TypeId))
6954 return TokError("expected type ID string for .amdgpu_typeid");
6955 ParsedInfoData.TypeIds.push_back({FuncSym, std::move(TypeId)});
6956 } else {
6957 return Error(IDLoc, "unknown .amdgpu_info directive '" + ID + "'");
6958 }
6959 }
6960
6961 if (HasScalarAttrs)
6962 ParsedInfoData.Funcs.push_back(std::move(FI));
6963
6964 AMDGPU::InfoSectionData &Data = InfoData ? *InfoData : InfoData.emplace();
6965 for (AMDGPU::FuncInfo &Func : ParsedInfoData.Funcs)
6966 Data.Funcs.push_back(std::move(Func));
6967 for (std::pair<MCSymbol *, MCSymbol *> &Use : ParsedInfoData.Uses)
6968 Data.Uses.push_back(Use);
6969 for (std::pair<MCSymbol *, MCSymbol *> &Call : ParsedInfoData.Calls)
6970 Data.Calls.push_back(Call);
6971 for (std::pair<MCSymbol *, std::string> &IndirectCall :
6972 ParsedInfoData.IndirectCalls)
6973 Data.IndirectCalls.push_back(std::move(IndirectCall));
6974 for (std::pair<MCSymbol *, std::string> &TypeId : ParsedInfoData.TypeIds)
6975 Data.TypeIds.push_back(std::move(TypeId));
6976
6977 return false;
6978}
6979
6980void AMDGPUAsmParser::doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) {
6981 // Record every parsed label in the timeline so that, at end of file, the
6982 // instructions following a kernel's label can be located regardless of
6983 // whether the .amdhsa_kernel directive came before or after the label.
6984 OpcodeStreamSymbols.emplace_back(Symbol, IDLoc, OpcodeStream.size());
6985}
6986
6987void AMDGPUAsmParser::checkKernelPrologues() {
6988 if (getFeatureBits()[AMDGPU::FeatureRequiresInitialUnclausedVmem]) {
6989 static const unsigned Required[] = {S_MOV_B64_gfx12, V_NOP_e32_gfx12,
6990 GLOBAL_PREFETCH_B8_SADDR_gfx1250};
6991 for (auto [Sym, Loc, Offset] : OpcodeStreamSymbols) {
6992 if (!AMDHSAKernelSymbols.contains(Sym))
6993 continue;
6994 ArrayRef<unsigned> Prologue = ArrayRef(OpcodeStream).drop_front(Offset);
6995 if (!Prologue.empty() && Prologue.front() == S_SETREG_IMM32_B32_gfx12)
6996 Prologue = Prologue.drop_front();
6997 if (Prologue.take_front(std::size(Required)) != ArrayRef(Required)) {
6998 Warning(Loc, "kernel '" + Sym->getName() +
6999 "' does not begin with the required prologue "
7000 "sequence: s_mov_b64 followed by v_nop and "
7001 "global_prefetch_b8");
7002 }
7003 }
7004 }
7005 OpcodeStream.clear();
7006 OpcodeStreamSymbols.clear();
7007 AMDHSAKernelSymbols.clear();
7008}
7009
7010void AMDGPUAsmParser::onEndOfFile() {
7011 emitTargetDirective();
7012 checkKernelPrologues();
7013 if (InfoData)
7014 getTargetStreamer().emitAMDGPUInfo(*InfoData);
7015}
7016
7017bool AMDGPUAsmParser::ParseDirective(AsmToken DirectiveID) {
7018 StringRef IDVal = DirectiveID.getString();
7019
7020 if (isHsaAbi(getSTI())) {
7021 if (IDVal == ".amdhsa_kernel")
7022 return ParseDirectiveAMDHSAKernel();
7023
7024 if (IDVal == ".amdhsa_code_object_version")
7025 return ParseDirectiveAMDHSACodeObjectVersion();
7026
7027 // TODO: Restructure/combine with PAL metadata directive.
7029 return ParseDirectiveHSAMetadata();
7030 } else {
7031 if (IDVal == ".amd_kernel_code_t")
7032 return ParseDirectiveAMDKernelCodeT();
7033
7034 if (IDVal == ".amdgpu_hsa_kernel")
7035 return ParseDirectiveAMDGPUHsaKernel();
7036
7037 if (IDVal == ".amd_amdgpu_isa")
7038 return ParseDirectiveISAVersion();
7039
7041 return Error(getLoc(), (Twine(HSAMD::AssemblerDirectiveBegin) +
7042 Twine(" directive is "
7043 "not available on non-amdhsa OSes"))
7044 .str());
7045 }
7046 }
7047
7048 if (IDVal == ".amdgcn_target")
7049 return ParseDirectiveAMDGCNTarget();
7050
7051 if (IDVal == ".amdgpu_lds")
7052 return ParseDirectiveAMDGPULDS();
7053
7054 if (IDVal == ".amdgpu_info")
7055 return ParseDirectiveAMDGPUInfo();
7056
7057 if (IDVal == PALMD::AssemblerDirectiveBegin)
7058 return ParseDirectivePALMetadataBegin();
7059
7060 if (IDVal == PALMD::AssemblerDirective)
7061 return ParseDirectivePALMetadata();
7062
7063 return true;
7064}
7065
7066bool AMDGPUAsmParser::subtargetHasRegister(const MCRegisterInfo &MRI,
7067 MCRegister Reg) {
7068 if (MRI.regsOverlap(TTMP12_TTMP13_TTMP14_TTMP15, Reg))
7069 return isGFX9Plus();
7070
7071 // GFX10+ has 2 more SGPRs 104 and 105.
7072 if (MRI.regsOverlap(SGPR104_SGPR105, Reg))
7073 return hasSGPR104_SGPR105();
7074
7075 switch (Reg.id()) {
7076 case SRC_SHARED_BASE_LO:
7077 case SRC_SHARED_BASE:
7078 case SRC_SHARED_LIMIT_LO:
7079 case SRC_SHARED_LIMIT:
7080 return isGFX9Plus();
7081 case SRC_PRIVATE_BASE_LO:
7082 case SRC_PRIVATE_BASE:
7083 case SRC_PRIVATE_LIMIT_LO:
7084 case SRC_PRIVATE_LIMIT:
7085 return AMDGPU::hasPrivateApertureRegs(getSTI());
7086 case SRC_FLAT_SCRATCH_BASE_LO:
7087 case SRC_FLAT_SCRATCH_BASE_HI:
7088 return hasGloballyAddressableScratch();
7089 case SRC_POPS_EXITING_WAVE_ID:
7090 return hasPopsExitingWaveID(getSTI());
7091 case TBA:
7092 case TBA_LO:
7093 case TBA_HI:
7094 case TMA:
7095 case TMA_LO:
7096 case TMA_HI:
7097 return !isGFX9Plus();
7098 case XNACK_MASK:
7099 case XNACK_MASK_LO:
7100 case XNACK_MASK_HI:
7101 return (isVI() || isGFX9()) &&
7102 getTargetStreamer().getTargetID()->isXnackSupported();
7103 case SGPR_NULL:
7104 return isGFX10Plus();
7105 case SRC_EXECZ:
7106 case SRC_VCCZ:
7107 return !isGFX11Plus();
7108 default:
7109 break;
7110 }
7111
7112 if (isCI())
7113 return true;
7114
7115 if (isSI() || isGFX10Plus()) {
7116 // No flat_scr on SI.
7117 // On GFX10Plus flat scratch is not a valid register operand and can only be
7118 // accessed with s_setreg/s_getreg.
7119 switch (Reg.id()) {
7120 case FLAT_SCR:
7121 case FLAT_SCR_LO:
7122 case FLAT_SCR_HI:
7123 return false;
7124 default:
7125 return true;
7126 }
7127 }
7128
7129 // VI only has 102 SGPRs, so make sure we aren't trying to use the 2 more that
7130 // SI/CI have.
7131 if (MRI.regsOverlap(SGPR102_SGPR103, Reg))
7132 return hasSGPR102_SGPR103();
7133
7134 return true;
7135}
7136
7137ParseStatus AMDGPUAsmParser::parseOperand(OperandVector &Operands,
7138 StringRef Mnemonic,
7139 OperandMode Mode) {
7140 ParseStatus Res = parseVOPD(Operands);
7141 if (Res.isSuccess() || Res.isFailure() || isToken(AsmToken::EndOfStatement))
7142 return Res;
7143
7144 // Try to parse with a custom parser
7145 Res = MatchOperandParserImpl(Operands, Mnemonic);
7146
7147 // If we successfully parsed the operand or if there as an error parsing,
7148 // we are done.
7149 //
7150 // If we are parsing after we reach EndOfStatement then this means we
7151 // are appending default values to the Operands list. This is only done
7152 // by custom parser, so we shouldn't continue on to the generic parsing.
7153 if (Res.isSuccess() || Res.isFailure() || isToken(AsmToken::EndOfStatement))
7154 return Res;
7155
7156 SMLoc RBraceLoc;
7157 SMLoc LBraceLoc = getLoc();
7158 if (Mode == OperandMode_NSA && trySkipToken(AsmToken::LBrac)) {
7159 unsigned Prefix = Operands.size();
7160
7161 for (;;) {
7162 auto Loc = getLoc();
7163 Res = parseReg(Operands);
7164 if (Res.isNoMatch())
7165 Error(Loc, "expected a register");
7166 if (!Res.isSuccess())
7167 return ParseStatus::Failure;
7168
7169 RBraceLoc = getLoc();
7170 if (trySkipToken(AsmToken::RBrac))
7171 break;
7172
7173 if (!skipToken(AsmToken::Comma,
7174 "expected a comma or a closing square bracket"))
7175 return ParseStatus::Failure;
7176 }
7177
7178 if (Operands.size() - Prefix > 1) {
7179 Operands.insert(Operands.begin() + Prefix,
7180 AMDGPUOperand::CreateToken(this, "[", LBraceLoc));
7181 Operands.push_back(AMDGPUOperand::CreateToken(this, "]", RBraceLoc));
7182 }
7183
7184 return ParseStatus::Success;
7185 }
7186
7187 return parseRegOrImm(Operands);
7188}
7189
7190StringRef AMDGPUAsmParser::parseMnemonicSuffix(StringRef Name) {
7191 // Clear any forced encodings from the previous instruction.
7192 setForcedEncodingSize(0);
7193 setForcedDPP(false);
7194 setForcedSDWA(false);
7195
7196 if (Name.consume_back("_e64_dpp")) {
7197 setForcedDPP(true);
7198 setForcedEncodingSize(64);
7199 return Name;
7200 }
7201 if (Name.consume_back("_e64")) {
7202 setForcedEncodingSize(64);
7203 return Name;
7204 }
7205 if (Name.consume_back("_e32")) {
7206 setForcedEncodingSize(32);
7207 return Name;
7208 }
7209 if (Name.consume_back("_dpp")) {
7210 setForcedDPP(true);
7211 return Name;
7212 }
7213 if (Name.consume_back("_sdwa")) {
7214 setForcedSDWA(true);
7215 return Name;
7216 }
7217 return Name;
7218}
7219
7220static void applyMnemonicAliases(StringRef &Mnemonic,
7221 const FeatureBitset &Features,
7222 unsigned VariantID);
7223
7224bool AMDGPUAsmParser::parseInstruction(ParseInstructionInfo &Info,
7225 StringRef Name, SMLoc NameLoc,
7227 // Add the instruction mnemonic
7228 Name = parseMnemonicSuffix(Name);
7229
7230 // If the target architecture uses MnemonicAlias, call it here to parse
7231 // operands correctly.
7232 applyMnemonicAliases(Name, getAvailableFeatures(), 0);
7233
7234 Operands.push_back(AMDGPUOperand::CreateToken(this, Name, NameLoc));
7235
7236 bool IsMIMG = Name.starts_with("image_");
7237
7238 while (!trySkipToken(AsmToken::EndOfStatement)) {
7239 OperandMode Mode = OperandMode_Default;
7240 if (IsMIMG && isGFX10Plus() && Operands.size() == 2)
7241 Mode = OperandMode_NSA;
7242 ParseStatus Res = parseOperand(Operands, Name, Mode);
7243
7244 if (!Res.isSuccess()) {
7245 checkUnsupportedInstruction(Name, NameLoc);
7246 if (!Parser.hasPendingError()) {
7247 // FIXME: use real operand location rather than the current location.
7248 StringRef Msg = Res.isFailure() ? "failed parsing operand."
7249 : "not a valid operand.";
7250 Error(getLoc(), Msg);
7251 }
7252 while (!trySkipToken(AsmToken::EndOfStatement)) {
7253 lex();
7254 }
7255 return true;
7256 }
7257
7258 // Eat the comma or space if there is one.
7259 trySkipToken(AsmToken::Comma);
7260 }
7261
7262 return false;
7263}
7264
7265//===----------------------------------------------------------------------===//
7266// Utility functions
7267//===----------------------------------------------------------------------===//
7268
7269ParseStatus AMDGPUAsmParser::parseTokenOp(StringRef Name,
7271 SMLoc S = getLoc();
7272 if (!trySkipId(Name))
7273 return ParseStatus::NoMatch;
7274
7275 Operands.push_back(AMDGPUOperand::CreateToken(this, Name, S));
7276 return ParseStatus::Success;
7277}
7278
7279ParseStatus AMDGPUAsmParser::parseIntWithPrefix(const char *Prefix,
7280 int64_t &IntVal) {
7281
7282 if (!trySkipId(Prefix, AsmToken::Colon))
7283 return ParseStatus::NoMatch;
7284
7286}
7287
7288ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
7289 const char *Prefix, OperandVector &Operands, AMDGPUOperand::ImmTy ImmTy,
7290 std::function<bool(int64_t &)> ConvertResult) {
7291 SMLoc S = getLoc();
7292 int64_t Value = 0;
7293
7294 ParseStatus Res = parseIntWithPrefix(Prefix, Value);
7295 if (!Res.isSuccess())
7296 return Res;
7297
7298 if (ConvertResult && !ConvertResult(Value)) {
7299 Error(S, "invalid " + StringRef(Prefix) + " value.");
7300 }
7301
7302 Operands.push_back(AMDGPUOperand::CreateImm(this, Value, S, ImmTy));
7303 return ParseStatus::Success;
7304}
7305
7306ParseStatus AMDGPUAsmParser::parseOperandArrayWithPrefix(
7307 const char *Prefix, OperandVector &Operands, AMDGPUOperand::ImmTy ImmTy,
7308 bool (*ConvertResult)(int64_t &)) {
7309 SMLoc S = getLoc();
7310 if (!trySkipId(Prefix, AsmToken::Colon))
7311 return ParseStatus::NoMatch;
7312
7313 if (!skipToken(AsmToken::LBrac, "expected a left square bracket"))
7314 return ParseStatus::Failure;
7315
7316 unsigned Val = 0;
7317 const unsigned MaxSize = 4;
7318
7319 // FIXME: How to verify the number of elements matches the number of src
7320 // operands?
7321 for (int I = 0;; ++I) {
7322 int64_t Op;
7323 SMLoc Loc = getLoc();
7324 if (!parseExpr(Op))
7325 return ParseStatus::Failure;
7326
7327 if (Op != 0 && Op != 1)
7328 return Error(Loc, "invalid " + StringRef(Prefix) + " value.");
7329
7330 Val |= (Op << I);
7331
7332 if (trySkipToken(AsmToken::RBrac))
7333 break;
7334
7335 if (I + 1 == MaxSize)
7336 return Error(getLoc(), "expected a closing square bracket");
7337
7338 if (!skipToken(AsmToken::Comma, "expected a comma"))
7339 return ParseStatus::Failure;
7340 }
7341
7342 Operands.push_back(AMDGPUOperand::CreateImm(this, Val, S, ImmTy));
7343 return ParseStatus::Success;
7344}
7345
7346ParseStatus AMDGPUAsmParser::parseNamedBit(StringRef Name,
7348 AMDGPUOperand::ImmTy ImmTy,
7349 bool IgnoreNegative) {
7350 int64_t Bit;
7351 SMLoc S = getLoc();
7352
7353 if (trySkipId(Name)) {
7354 Bit = 1;
7355 } else if (trySkipId("no", Name)) {
7356 if (IgnoreNegative)
7357 return ParseStatus::Success;
7358 Bit = 0;
7359 } else {
7360 return ParseStatus::NoMatch;
7361 }
7362
7363 if (Name == "r128" && !hasMIMG_R128())
7364 return Error(S, "r128 modifier is not supported on this GPU");
7365 if (Name == "a16" && !hasA16())
7366 return Error(S, "a16 modifier is not supported on this GPU");
7367
7368 if (Bit == 0 && Name == "gds") {
7369 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
7370 if (Mnemo.starts_with("ds_gws"))
7371 return Error(S, "nogds is not allowed");
7372 }
7373
7374 if (isGFX9() && ImmTy == AMDGPUOperand::ImmTyA16)
7375 ImmTy = AMDGPUOperand::ImmTyR128A16;
7376
7377 Operands.push_back(AMDGPUOperand::CreateImm(this, Bit, S, ImmTy));
7378 return ParseStatus::Success;
7379}
7380
7381unsigned AMDGPUAsmParser::getCPolKind(StringRef Id, StringRef Mnemo,
7382 bool &Disabling) const {
7383 Disabling = Id.consume_front("no");
7384
7385 if (isGFX940() && !Mnemo.starts_with("s_")) {
7386 return StringSwitch<unsigned>(Id)
7387 .Case("nt", AMDGPU::CPol::NT)
7388 .Case("sc0", AMDGPU::CPol::SC0)
7389 .Case("sc1", AMDGPU::CPol::SC1)
7390 .Default(0);
7391 }
7392
7393 return StringSwitch<unsigned>(Id)
7394 .Case("dlc", AMDGPU::CPol::DLC)
7395 .Case("glc", AMDGPU::CPol::GLC)
7396 .Case("scc", AMDGPU::CPol::SCC)
7397 .Case("slc", AMDGPU::CPol::SLC)
7398 .Default(0);
7399}
7400
7401ParseStatus AMDGPUAsmParser::parseCPol(OperandVector &Operands) {
7402 if (isGFX12Plus()) {
7403 SMLoc StringLoc = getLoc();
7404
7405 int64_t CPolVal = 0;
7406 ParseStatus ResTH = ParseStatus::NoMatch;
7407 ParseStatus ResScope = ParseStatus::NoMatch;
7408 ParseStatus ResNV = ParseStatus::NoMatch;
7409 ParseStatus ResScal = ParseStatus::NoMatch;
7410
7411 for (;;) {
7412 if (ResTH.isNoMatch()) {
7413 int64_t TH;
7414 ResTH = parseTH(Operands, TH);
7415 if (ResTH.isFailure())
7416 return ResTH;
7417 if (ResTH.isSuccess()) {
7418 CPolVal |= TH;
7419 continue;
7420 }
7421 }
7422
7423 if (ResScope.isNoMatch()) {
7424 int64_t Scope;
7425 ResScope = parseScope(Operands, Scope);
7426 if (ResScope.isFailure())
7427 return ResScope;
7428 if (ResScope.isSuccess()) {
7429 CPolVal |= Scope;
7430 continue;
7431 }
7432 }
7433
7434 // NV bit exists on GFX12+, but does something starting from GFX1250.
7435 // Allow parsing on all GFX12 and fail on validation for better
7436 // diagnostics.
7437 if (ResNV.isNoMatch()) {
7438 if (trySkipId("nv")) {
7439 ResNV = ParseStatus::Success;
7440 CPolVal |= CPol::NV;
7441 continue;
7442 } else if (trySkipId("no", "nv")) {
7443 ResNV = ParseStatus::Success;
7444 continue;
7445 }
7446 }
7447
7448 if (ResScal.isNoMatch()) {
7449 if (trySkipId("scale_offset")) {
7450 ResScal = ParseStatus::Success;
7451 CPolVal |= CPol::SCAL;
7452 continue;
7453 } else if (trySkipId("no", "scale_offset")) {
7454 ResScal = ParseStatus::Success;
7455 continue;
7456 }
7457 }
7458
7459 break;
7460 }
7461
7462 if (ResTH.isNoMatch() && ResScope.isNoMatch() && ResNV.isNoMatch() &&
7463 ResScal.isNoMatch())
7464 return ParseStatus::NoMatch;
7465
7466 Operands.push_back(AMDGPUOperand::CreateImm(this, CPolVal, StringLoc,
7467 AMDGPUOperand::ImmTyCPol));
7468 return ParseStatus::Success;
7469 }
7470
7471 StringRef Mnemo = ((AMDGPUOperand &)*Operands[0]).getToken();
7472 SMLoc OpLoc = getLoc();
7473 unsigned Enabled = 0, Seen = 0;
7474 for (;;) {
7475 SMLoc S = getLoc();
7476 bool Disabling;
7477 unsigned CPol = getCPolKind(getId(), Mnemo, Disabling);
7478 if (!CPol)
7479 break;
7480
7481 lex();
7482
7483 if (!isGFX10Plus() && CPol == AMDGPU::CPol::DLC)
7484 return Error(S, "dlc modifier is not supported on this GPU");
7485
7486 if (!isGFX90A() && CPol == AMDGPU::CPol::SCC)
7487 return Error(S, "scc modifier is not supported on this GPU");
7488
7489 if (Seen & CPol)
7490 return Error(S, "duplicate cache policy modifier");
7491
7492 if (!Disabling)
7493 Enabled |= CPol;
7494
7495 Seen |= CPol;
7496 }
7497
7498 if (!Seen)
7499 return ParseStatus::NoMatch;
7500
7501 Operands.push_back(
7502 AMDGPUOperand::CreateImm(this, Enabled, OpLoc, AMDGPUOperand::ImmTyCPol));
7503 return ParseStatus::Success;
7504}
7505
7506ParseStatus AMDGPUAsmParser::parseScope(OperandVector &Operands,
7507 int64_t &Scope) {
7508 static const unsigned Scopes[] = {CPol::SCOPE_CU, CPol::SCOPE_SE,
7510
7511 ParseStatus Res = parseStringOrIntWithPrefix(
7512 Operands, "scope", {"SCOPE_CU", "SCOPE_SE", "SCOPE_DEV", "SCOPE_SYS"},
7513 Scope);
7514
7515 if (Res.isSuccess())
7516 Scope = Scopes[Scope];
7517
7518 return Res;
7519}
7520
7521ParseStatus AMDGPUAsmParser::parseTH(OperandVector &Operands, int64_t &TH) {
7522 TH = AMDGPU::CPol::TH_RT; // default
7523
7524 StringRef Value;
7525 SMLoc StringLoc;
7526 ParseStatus Res = parseStringWithPrefix("th", Value, StringLoc);
7527 if (!Res.isSuccess())
7528 return Res;
7529
7530 if (Value == "TH_DEFAULT")
7532 else if (Value == "TH_STORE_LU" || Value == "TH_LOAD_WB" ||
7533 Value == "TH_LOAD_NT_WB") {
7534 return Error(StringLoc, "invalid th value");
7535 } else if (Value.consume_front("TH_ATOMIC_")) {
7537 } else if (Value.consume_front("TH_LOAD_")) {
7539 } else if (Value.consume_front("TH_STORE_")) {
7541 } else {
7542 return Error(StringLoc, "invalid th value");
7543 }
7544
7545 if (Value == "BYPASS")
7547
7548 if (TH != 0) {
7550 TH |= StringSwitch<int64_t>(Value)
7551 .Case("RETURN", AMDGPU::CPol::TH_ATOMIC_RETURN)
7552 .Case("RT", AMDGPU::CPol::TH_RT)
7553 .Case("RT_RETURN", AMDGPU::CPol::TH_ATOMIC_RETURN)
7554 .Case("NT", AMDGPU::CPol::TH_ATOMIC_NT)
7555 .Case("NT_RETURN", AMDGPU::CPol::TH_ATOMIC_NT |
7557 .Case("CASCADE_RT", AMDGPU::CPol::TH_ATOMIC_CASCADE)
7558 .Case("CASCADE_NT", AMDGPU::CPol::TH_ATOMIC_CASCADE |
7560 .Default(0xffffffff);
7561 else
7562 TH |= StringSwitch<int64_t>(Value)
7563 .Case("RT", AMDGPU::CPol::TH_RT)
7564 .Case("NT", AMDGPU::CPol::TH_NT)
7565 .Case("HT", AMDGPU::CPol::TH_HT)
7566 .Case("LU", AMDGPU::CPol::TH_LU)
7567 .Case("WB", AMDGPU::CPol::TH_WB)
7568 .Case("NT_RT", AMDGPU::CPol::TH_NT_RT)
7569 .Case("RT_NT", AMDGPU::CPol::TH_RT_NT)
7570 .Case("NT_HT", AMDGPU::CPol::TH_NT_HT)
7571 .Case("NT_WB", AMDGPU::CPol::TH_NT_WB)
7572 .Case("BYPASS", AMDGPU::CPol::TH_BYPASS)
7573 .Default(0xffffffff);
7574 }
7575
7576 if (TH == 0xffffffff)
7577 return Error(StringLoc, "invalid th value");
7578
7579 return ParseStatus::Success;
7580}
7581
7582static void
7584 AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx,
7585 AMDGPUOperand::ImmTy ImmT, int64_t Default = 0,
7586 std::optional<unsigned> InsertAt = std::nullopt) {
7587 auto i = OptionalIdx.find(ImmT);
7588 if (i != OptionalIdx.end()) {
7589 unsigned Idx = i->second;
7590 const AMDGPUOperand &Op =
7591 static_cast<const AMDGPUOperand &>(*Operands[Idx]);
7592 if (InsertAt)
7593 Inst.insert(Inst.begin() + *InsertAt, MCOperand::createImm(Op.getImm()));
7594 else
7595 Op.addImmOperands(Inst, 1);
7596 } else {
7597 if (InsertAt.has_value())
7598 Inst.insert(Inst.begin() + *InsertAt, MCOperand::createImm(Default));
7599 else
7601 }
7602}
7603
7604ParseStatus AMDGPUAsmParser::parseStringWithPrefix(StringRef Prefix,
7605 StringRef &Value,
7606 SMLoc &StringLoc) {
7607 if (!trySkipId(Prefix, AsmToken::Colon))
7608 return ParseStatus::NoMatch;
7609
7610 StringLoc = getLoc();
7611 return parseId(Value, "expected an identifier") ? ParseStatus::Success
7613}
7614
7615ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7616 OperandVector &Operands, StringRef Name, ArrayRef<const char *> Ids,
7617 int64_t &IntVal) {
7618 if (!trySkipId(Name, AsmToken::Colon))
7619 return ParseStatus::NoMatch;
7620
7621 SMLoc StringLoc = getLoc();
7622
7623 StringRef Value;
7624 if (isToken(AsmToken::Identifier)) {
7625 Value = getTokenStr();
7626 lex();
7627
7628 for (IntVal = 0; IntVal < (int64_t)Ids.size(); ++IntVal)
7629 if (Value == Ids[IntVal])
7630 break;
7631 } else if (!parseExpr(IntVal))
7632 return ParseStatus::Failure;
7633
7634 if (IntVal < 0 || IntVal >= (int64_t)Ids.size())
7635 return Error(StringLoc, "invalid " + Twine(Name) + " value");
7636
7637 return ParseStatus::Success;
7638}
7639
7640ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7641 OperandVector &Operands, StringRef Name, ArrayRef<const char *> Ids,
7642 AMDGPUOperand::ImmTy Type) {
7643 SMLoc S = getLoc();
7644 int64_t IntVal;
7645
7646 ParseStatus Res = parseStringOrIntWithPrefix(Operands, Name, Ids, IntVal);
7647 if (Res.isSuccess())
7648 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S, Type));
7649
7650 return Res;
7651}
7652
7653//===----------------------------------------------------------------------===//
7654// MTBUF format
7655//===----------------------------------------------------------------------===//
7656
7657bool AMDGPUAsmParser::tryParseFmt(const char *Pref, int64_t MaxVal,
7658 int64_t &Fmt) {
7659 int64_t Val;
7660 SMLoc Loc = getLoc();
7661
7662 auto Res = parseIntWithPrefix(Pref, Val);
7663 if (Res.isFailure())
7664 return false;
7665 if (Res.isNoMatch())
7666 return true;
7667
7668 if (Val < 0 || Val > MaxVal) {
7669 Error(Loc, Twine("out of range ", StringRef(Pref)));
7670 return false;
7671 }
7672
7673 Fmt = Val;
7674 return true;
7675}
7676
7677ParseStatus AMDGPUAsmParser::tryParseIndexKey(OperandVector &Operands,
7678 AMDGPUOperand::ImmTy ImmTy) {
7679 const char *Pref = "index_key";
7680 int64_t ImmVal = 0;
7681 SMLoc Loc = getLoc();
7682 auto Res = parseIntWithPrefix(Pref, ImmVal);
7683 if (!Res.isSuccess())
7684 return Res;
7685
7686 if ((ImmTy == AMDGPUOperand::ImmTyIndexKey16bit ||
7687 ImmTy == AMDGPUOperand::ImmTyIndexKey32bit) &&
7688 (ImmVal < 0 || ImmVal > 1))
7689 return Error(Loc, Twine("out of range ", StringRef(Pref)));
7690
7691 if (ImmTy == AMDGPUOperand::ImmTyIndexKey8bit && (ImmVal < 0 || ImmVal > 3))
7692 return Error(Loc, Twine("out of range ", StringRef(Pref)));
7693
7694 Operands.push_back(AMDGPUOperand::CreateImm(this, ImmVal, Loc, ImmTy));
7695 return ParseStatus::Success;
7696}
7697
7698ParseStatus AMDGPUAsmParser::parseIndexKey8bit(OperandVector &Operands) {
7699 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey8bit);
7700}
7701
7702ParseStatus AMDGPUAsmParser::parseIndexKey16bit(OperandVector &Operands) {
7703 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey16bit);
7704}
7705
7706ParseStatus AMDGPUAsmParser::parseIndexKey32bit(OperandVector &Operands) {
7707 return tryParseIndexKey(Operands, AMDGPUOperand::ImmTyIndexKey32bit);
7708}
7709
7710ParseStatus AMDGPUAsmParser::tryParseMatrixFMT(OperandVector &Operands,
7711 StringRef Name,
7712 AMDGPUOperand::ImmTy Type) {
7713 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixFmt,
7714 Type);
7715}
7716
7717ParseStatus AMDGPUAsmParser::parseMatrixAFMT(OperandVector &Operands) {
7718 return tryParseMatrixFMT(Operands, "matrix_a_fmt",
7719 AMDGPUOperand::ImmTyMatrixAFMT);
7720}
7721
7722ParseStatus AMDGPUAsmParser::parseMatrixBFMT(OperandVector &Operands) {
7723 return tryParseMatrixFMT(Operands, "matrix_b_fmt",
7724 AMDGPUOperand::ImmTyMatrixBFMT);
7725}
7726
7727ParseStatus AMDGPUAsmParser::tryParseMatrixScale(OperandVector &Operands,
7728 StringRef Name,
7729 AMDGPUOperand::ImmTy Type) {
7730 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixScale,
7731 Type);
7732}
7733
7734ParseStatus AMDGPUAsmParser::parseMatrixAScale(OperandVector &Operands) {
7735 return tryParseMatrixScale(Operands, "matrix_a_scale",
7736 AMDGPUOperand::ImmTyMatrixAScale);
7737}
7738
7739ParseStatus AMDGPUAsmParser::parseMatrixBScale(OperandVector &Operands) {
7740 return tryParseMatrixScale(Operands, "matrix_b_scale",
7741 AMDGPUOperand::ImmTyMatrixBScale);
7742}
7743
7744ParseStatus AMDGPUAsmParser::tryParseMatrixScaleFmt(OperandVector &Operands,
7745 StringRef Name,
7746 AMDGPUOperand::ImmTy Type) {
7747 return parseStringOrIntWithPrefix(Operands, Name, WMMAMods::ModMatrixScaleFmt,
7748 Type);
7749}
7750
7751ParseStatus AMDGPUAsmParser::parseMatrixAScaleFmt(OperandVector &Operands) {
7752 return tryParseMatrixScaleFmt(Operands, "matrix_a_scale_fmt",
7753 AMDGPUOperand::ImmTyMatrixAScaleFmt);
7754}
7755
7756ParseStatus AMDGPUAsmParser::parseMatrixBScaleFmt(OperandVector &Operands) {
7757 return tryParseMatrixScaleFmt(Operands, "matrix_b_scale_fmt",
7758 AMDGPUOperand::ImmTyMatrixBScaleFmt);
7759}
7760
7761// dfmt and nfmt (in a tbuffer instruction) are parsed as one to allow their
7762// values to live in a joint format operand in the MCInst encoding.
7763ParseStatus AMDGPUAsmParser::parseDfmtNfmt(int64_t &Format) {
7764 using namespace llvm::AMDGPU::MTBUFFormat;
7765
7766 int64_t Dfmt = DFMT_UNDEF;
7767 int64_t Nfmt = NFMT_UNDEF;
7768
7769 // dfmt and nfmt can appear in either order, and each is optional.
7770 for (int I = 0; I < 2; ++I) {
7771 if (Dfmt == DFMT_UNDEF && !tryParseFmt("dfmt", DFMT_MAX, Dfmt))
7772 return ParseStatus::Failure;
7773
7774 if (Nfmt == NFMT_UNDEF && !tryParseFmt("nfmt", NFMT_MAX, Nfmt))
7775 return ParseStatus::Failure;
7776
7777 // Skip optional comma between dfmt/nfmt
7778 // but guard against 2 commas following each other.
7779 if ((Dfmt == DFMT_UNDEF) != (Nfmt == NFMT_UNDEF) &&
7780 !peekToken().is(AsmToken::Comma)) {
7781 trySkipToken(AsmToken::Comma);
7782 }
7783 }
7784
7785 if (Dfmt == DFMT_UNDEF && Nfmt == NFMT_UNDEF)
7786 return ParseStatus::NoMatch;
7787
7788 Dfmt = (Dfmt == DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
7789 Nfmt = (Nfmt == NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
7790
7791 Format = encodeDfmtNfmt(Dfmt, Nfmt);
7792 return ParseStatus::Success;
7793}
7794
7795ParseStatus AMDGPUAsmParser::parseUfmt(int64_t &Format) {
7796 using namespace llvm::AMDGPU::MTBUFFormat;
7797
7798 int64_t Fmt = UFMT_UNDEF;
7799
7800 if (!tryParseFmt("format", UFMT_MAX, Fmt))
7801 return ParseStatus::Failure;
7802
7803 if (Fmt == UFMT_UNDEF)
7804 return ParseStatus::NoMatch;
7805
7806 Format = Fmt;
7807 return ParseStatus::Success;
7808}
7809
7810bool AMDGPUAsmParser::matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt,
7811 StringRef FormatStr, SMLoc Loc) {
7812 using namespace llvm::AMDGPU::MTBUFFormat;
7813 int64_t Format;
7814
7815 Format = getDfmt(FormatStr);
7816 if (Format != DFMT_UNDEF) {
7817 Dfmt = Format;
7818 return true;
7819 }
7820
7821 Format = getNfmt(FormatStr, getSTI());
7822 if (Format != NFMT_UNDEF) {
7823 Nfmt = Format;
7824 return true;
7825 }
7826
7827 Error(Loc, "unsupported format");
7828 return false;
7829}
7830
7831ParseStatus AMDGPUAsmParser::parseSymbolicSplitFormat(StringRef FormatStr,
7832 SMLoc FormatLoc,
7833 int64_t &Format) {
7834 using namespace llvm::AMDGPU::MTBUFFormat;
7835
7836 int64_t Dfmt = DFMT_UNDEF;
7837 int64_t Nfmt = NFMT_UNDEF;
7838 if (!matchDfmtNfmt(Dfmt, Nfmt, FormatStr, FormatLoc))
7839 return ParseStatus::Failure;
7840
7841 if (trySkipToken(AsmToken::Comma)) {
7842 StringRef Str;
7843 SMLoc Loc = getLoc();
7844 if (!parseId(Str, "expected a format string") ||
7845 !matchDfmtNfmt(Dfmt, Nfmt, Str, Loc))
7846 return ParseStatus::Failure;
7847 if (Dfmt == DFMT_UNDEF)
7848 return Error(Loc, "duplicate numeric format");
7849 if (Nfmt == NFMT_UNDEF)
7850 return Error(Loc, "duplicate data format");
7851 }
7852
7853 Dfmt = (Dfmt == DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
7854 Nfmt = (Nfmt == NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
7855
7856 if (isGFX10Plus()) {
7857 auto Ufmt = convertDfmtNfmt2Ufmt(Dfmt, Nfmt, getSTI());
7858 if (Ufmt == UFMT_UNDEF)
7859 return Error(FormatLoc, "unsupported format");
7860 Format = Ufmt;
7861 } else {
7862 Format = encodeDfmtNfmt(Dfmt, Nfmt);
7863 }
7864
7865 return ParseStatus::Success;
7866}
7867
7868ParseStatus AMDGPUAsmParser::parseSymbolicUnifiedFormat(StringRef FormatStr,
7869 SMLoc Loc,
7870 int64_t &Format) {
7871 using namespace llvm::AMDGPU::MTBUFFormat;
7872
7873 auto Id = getUnifiedFormat(FormatStr, getSTI());
7874 if (Id == UFMT_UNDEF)
7875 return ParseStatus::NoMatch;
7876
7877 if (!isGFX10Plus())
7878 return Error(Loc, "unified format is not supported on this GPU");
7879
7880 Format = Id;
7881 return ParseStatus::Success;
7882}
7883
7884ParseStatus AMDGPUAsmParser::parseNumericFormat(int64_t &Format) {
7885 using namespace llvm::AMDGPU::MTBUFFormat;
7886 SMLoc Loc = getLoc();
7887
7888 if (!parseExpr(Format))
7889 return ParseStatus::Failure;
7890 if (!isValidFormatEncoding(Format, getSTI()))
7891 return Error(Loc, "out of range format");
7892
7893 return ParseStatus::Success;
7894}
7895
7896ParseStatus AMDGPUAsmParser::parseSymbolicOrNumericFormat(int64_t &Format) {
7897 using namespace llvm::AMDGPU::MTBUFFormat;
7898
7899 if (!trySkipId("format", AsmToken::Colon))
7900 return ParseStatus::NoMatch;
7901
7902 if (trySkipToken(AsmToken::LBrac)) {
7903 StringRef FormatStr;
7904 SMLoc Loc = getLoc();
7905 if (!parseId(FormatStr, "expected a format string"))
7906 return ParseStatus::Failure;
7907
7908 auto Res = parseSymbolicUnifiedFormat(FormatStr, Loc, Format);
7909 if (Res.isNoMatch())
7910 Res = parseSymbolicSplitFormat(FormatStr, Loc, Format);
7911 if (!Res.isSuccess())
7912 return Res;
7913
7914 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
7915 return ParseStatus::Failure;
7916
7917 return ParseStatus::Success;
7918 }
7919
7920 return parseNumericFormat(Format);
7921}
7922
7923ParseStatus AMDGPUAsmParser::parseFORMAT(OperandVector &Operands) {
7924 using namespace llvm::AMDGPU::MTBUFFormat;
7925
7926 int64_t Format = getDefaultFormatEncoding(getSTI());
7927 ParseStatus Res;
7928 SMLoc Loc = getLoc();
7929
7930 // Parse legacy format syntax.
7931 Res = isGFX10Plus() ? parseUfmt(Format) : parseDfmtNfmt(Format);
7932 if (Res.isFailure())
7933 return Res;
7934
7935 bool FormatFound = Res.isSuccess();
7936
7937 Operands.push_back(
7938 AMDGPUOperand::CreateImm(this, Format, Loc, AMDGPUOperand::ImmTyFORMAT));
7939
7940 if (FormatFound)
7941 trySkipToken(AsmToken::Comma);
7942
7943 if (isToken(AsmToken::EndOfStatement)) {
7944 // We are expecting an soffset operand,
7945 // but let matcher handle the error.
7946 return ParseStatus::Success;
7947 }
7948
7949 // Parse soffset.
7950 Res = parseRegOrImm(Operands);
7951 if (!Res.isSuccess())
7952 return Res;
7953
7954 trySkipToken(AsmToken::Comma);
7955
7956 if (!FormatFound) {
7957 Res = parseSymbolicOrNumericFormat(Format);
7958 if (Res.isFailure())
7959 return Res;
7960 if (Res.isSuccess()) {
7961 auto Size = Operands.size();
7962 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands[Size - 2]);
7963 assert(Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyFORMAT);
7964 Op.setImm(Format);
7965 }
7966 return ParseStatus::Success;
7967 }
7968
7969 if (isId("format") && peekToken().is(AsmToken::Colon))
7970 return Error(getLoc(), "duplicate format");
7971 return ParseStatus::Success;
7972}
7973
7974ParseStatus AMDGPUAsmParser::parseFlatOffset(OperandVector &Operands) {
7975 ParseStatus Res =
7976 parseIntWithPrefix("offset", Operands, AMDGPUOperand::ImmTyOffset);
7977 if (Res.isNoMatch()) {
7978 Res = parseIntWithPrefix("inst_offset", Operands,
7979 AMDGPUOperand::ImmTyInstOffset);
7980 }
7981 return Res;
7982}
7983
7984ParseStatus AMDGPUAsmParser::parseR128A16(OperandVector &Operands) {
7985 ParseStatus Res =
7986 parseNamedBit("r128", Operands, AMDGPUOperand::ImmTyR128A16);
7987 if (Res.isNoMatch())
7988 Res = parseNamedBit("a16", Operands, AMDGPUOperand::ImmTyA16);
7989 return Res;
7990}
7991
7992ParseStatus AMDGPUAsmParser::parseBLGP(OperandVector &Operands) {
7993 ParseStatus Res =
7994 parseIntWithPrefix("blgp", Operands, AMDGPUOperand::ImmTyBLGP);
7995 if (Res.isNoMatch()) {
7996 Res =
7997 parseOperandArrayWithPrefix("neg", Operands, AMDGPUOperand::ImmTyBLGP);
7998 }
7999 return Res;
8000}
8001
8002//===----------------------------------------------------------------------===//
8003// Exp
8004//===----------------------------------------------------------------------===//
8005
8006void AMDGPUAsmParser::cvtExp(MCInst &Inst, const OperandVector &Operands) {
8007 OptionalImmIndexMap OptionalIdx;
8008
8009 unsigned OperandIdx[4];
8010 unsigned EnMask = 0;
8011 int SrcIdx = 0;
8012
8013 for (unsigned i = 1, e = Operands.size(); i != e; ++i) {
8014 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
8015
8016 // Add the register arguments
8017 if (Op.isReg()) {
8018 assert(SrcIdx < 4);
8019 OperandIdx[SrcIdx] = Inst.size();
8020 Op.addRegOperands(Inst, 1);
8021 ++SrcIdx;
8022 continue;
8023 }
8024
8025 if (Op.isOff()) {
8026 assert(SrcIdx < 4);
8027 OperandIdx[SrcIdx] = Inst.size();
8028 Inst.addOperand(MCOperand::createReg(MCRegister()));
8029 ++SrcIdx;
8030 continue;
8031 }
8032
8033 if (Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyExpTgt) {
8034 Op.addImmOperands(Inst, 1);
8035 continue;
8036 }
8037
8038 if (Op.isToken() && (Op.getToken() == "done" || Op.getToken() == "row_en"))
8039 continue;
8040
8041 // Handle optional arguments
8042 OptionalIdx[Op.getImmTy()] = i;
8043 }
8044
8045 assert(SrcIdx == 4);
8046
8047 bool Compr = false;
8048 if (OptionalIdx.find(AMDGPUOperand::ImmTyExpCompr) != OptionalIdx.end()) {
8049 Compr = true;
8050 Inst.getOperand(OperandIdx[1]) = Inst.getOperand(OperandIdx[2]);
8051 Inst.getOperand(OperandIdx[2]).setReg(MCRegister());
8052 Inst.getOperand(OperandIdx[3]).setReg(MCRegister());
8053 }
8054
8055 for (auto i = 0; i < SrcIdx; ++i) {
8056 if (Inst.getOperand(OperandIdx[i]).getReg()) {
8057 EnMask |= Compr ? (0x3 << i * 2) : (0x1 << i);
8058 }
8059 }
8060
8061 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyExpVM);
8062 addOptionalImmOperand(Inst, Operands, OptionalIdx,
8063 AMDGPUOperand::ImmTyExpCompr);
8064
8065 Inst.addOperand(MCOperand::createImm(EnMask));
8066}
8067
8068//===----------------------------------------------------------------------===//
8069// s_waitcnt
8070//===----------------------------------------------------------------------===//
8071
8072static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal,
8073 int64_t CntVal, bool Saturate,
8074 unsigned (*encode)(const IsaVersion &Version, unsigned,
8075 unsigned),
8076 unsigned (*decode)(const IsaVersion &Version, unsigned)) {
8077 bool Failed = false;
8078
8079 IntVal = encode(ISA, IntVal, CntVal);
8080 if (CntVal != decode(ISA, IntVal)) {
8081 if (Saturate) {
8082 IntVal = encode(ISA, IntVal, -1);
8083 } else {
8084 Failed = true;
8085 }
8086 }
8087 return Failed;
8088}
8089
8090bool AMDGPUAsmParser::parseCnt(int64_t &IntVal) {
8091
8092 SMLoc CntLoc = getLoc();
8093 StringRef CntName = getTokenStr();
8094
8095 if (!skipToken(AsmToken::Identifier, "expected a counter name") ||
8096 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8097 return false;
8098
8099 int64_t CntVal;
8100 SMLoc ValLoc = getLoc();
8101 if (!parseExpr(CntVal))
8102 return false;
8103
8104 bool Failed = true;
8105 bool Sat = CntName.ends_with("_sat");
8106
8107 if (CntName == "vmcnt" || CntName == "vmcnt_sat") {
8108 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeVmcnt, decodeVmcnt);
8109 } else if (CntName == "expcnt" || CntName == "expcnt_sat") {
8110 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeExpcnt, decodeExpcnt);
8111 } else if (CntName == "lgkmcnt" || CntName == "lgkmcnt_sat") {
8112 Failed = encodeCnt(ISA, IntVal, CntVal, Sat, encodeLgkmcnt, decodeLgkmcnt);
8113 } else {
8114 Error(CntLoc, "invalid counter name " + CntName);
8115 return false;
8116 }
8117
8118 if (Failed) {
8119 Error(ValLoc, "too large value for " + CntName);
8120 return false;
8121 }
8122
8123 if (!skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8124 return false;
8125
8126 if (trySkipToken(AsmToken::Amp) || trySkipToken(AsmToken::Comma)) {
8127 if (isToken(AsmToken::EndOfStatement)) {
8128 Error(getLoc(), "expected a counter name");
8129 return false;
8130 }
8131 }
8132
8133 return true;
8134}
8135
8136ParseStatus AMDGPUAsmParser::parseSWaitCnt(OperandVector &Operands) {
8137 int64_t Waitcnt = getWaitcntBitMask(ISA);
8138 SMLoc S = getLoc();
8139
8140 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8141 while (!isToken(AsmToken::EndOfStatement)) {
8142 if (!parseCnt(Waitcnt))
8143 return ParseStatus::Failure;
8144 }
8145 } else {
8146 if (!parseExpr(Waitcnt))
8147 return ParseStatus::Failure;
8148 }
8149
8150 Operands.push_back(AMDGPUOperand::CreateImm(this, Waitcnt, S));
8151 return ParseStatus::Success;
8152}
8153
8154bool AMDGPUAsmParser::parseDelay(int64_t &Delay) {
8155 SMLoc FieldLoc = getLoc();
8156 StringRef FieldName = getTokenStr();
8157 if (!skipToken(AsmToken::Identifier, "expected a field name") ||
8158 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8159 return false;
8160
8161 SMLoc ValueLoc = getLoc();
8162 StringRef ValueName = getTokenStr();
8163 if (!skipToken(AsmToken::Identifier, "expected a value name") ||
8164 !skipToken(AsmToken::RParen, "expected a right parenthesis"))
8165 return false;
8166
8167 unsigned Shift;
8168 if (FieldName == "instid0") {
8169 Shift = 0;
8170 } else if (FieldName == "instskip") {
8171 Shift = 4;
8172 } else if (FieldName == "instid1") {
8173 Shift = 7;
8174 } else {
8175 Error(FieldLoc, "invalid field name " + FieldName);
8176 return false;
8177 }
8178
8179 int Value;
8180 if (Shift == 4) {
8181 // Parse values for instskip.
8182 Value = StringSwitch<int>(ValueName)
8183 .Case("SAME", 0)
8184 .Case("NEXT", 1)
8185 .Case("SKIP_1", 2)
8186 .Case("SKIP_2", 3)
8187 .Case("SKIP_3", 4)
8188 .Case("SKIP_4", 5)
8189 .Default(-1);
8190 } else {
8191 // Parse values for instid0 and instid1.
8192 Value = StringSwitch<int>(ValueName)
8193 .Case("NO_DEP", 0)
8194 .Case("VALU_DEP_1", 1)
8195 .Case("VALU_DEP_2", 2)
8196 .Case("VALU_DEP_3", 3)
8197 .Case("VALU_DEP_4", 4)
8198 .Case("TRANS32_DEP_1", 5)
8199 .Case("TRANS32_DEP_2", 6)
8200 .Case("TRANS32_DEP_3", 7)
8201 .Case("FMA_ACCUM_CYCLE_1", 8)
8202 .Case("SALU_CYCLE_1", 9)
8203 .Case("SALU_CYCLE_2", 10)
8204 .Case("SALU_CYCLE_3", 11)
8205 .Default(-1);
8206 }
8207 if (Value < 0) {
8208 Error(ValueLoc, "invalid value name " + ValueName);
8209 return false;
8210 }
8211
8212 Delay |= Value << Shift;
8213 return true;
8214}
8215
8216ParseStatus AMDGPUAsmParser::parseSDelayALU(OperandVector &Operands) {
8217 int64_t Delay = 0;
8218 SMLoc S = getLoc();
8219
8220 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8221 do {
8222 if (!parseDelay(Delay))
8223 return ParseStatus::Failure;
8224 } while (trySkipToken(AsmToken::Pipe));
8225 } else {
8226 if (!parseExpr(Delay))
8227 return ParseStatus::Failure;
8228 }
8229
8230 Operands.push_back(AMDGPUOperand::CreateImm(this, Delay, S));
8231 return ParseStatus::Success;
8232}
8233
8234bool AMDGPUOperand::isSWaitCnt() const { return isImm(); }
8235
8236bool AMDGPUOperand::isSDelayALU() const { return isImm(); }
8237
8238//===----------------------------------------------------------------------===//
8239// DepCtr
8240//===----------------------------------------------------------------------===//
8241
8242void AMDGPUAsmParser::depCtrError(SMLoc Loc, int ErrorId,
8243 StringRef DepCtrName) {
8244 switch (ErrorId) {
8245 case OPR_ID_UNKNOWN:
8246 Error(Loc, Twine("invalid counter name ", DepCtrName));
8247 return;
8248 case OPR_ID_UNSUPPORTED:
8249 Error(Loc, Twine(DepCtrName, " is not supported on this GPU"));
8250 return;
8251 case OPR_ID_DUPLICATE:
8252 Error(Loc, Twine("duplicate counter name ", DepCtrName));
8253 return;
8254 case OPR_VAL_INVALID:
8255 Error(Loc, Twine("invalid value for ", DepCtrName));
8256 return;
8257 default:
8258 assert(false);
8259 }
8260}
8261
8262bool AMDGPUAsmParser::parseDepCtr(int64_t &DepCtr, unsigned &UsedOprMask) {
8263
8264 using namespace llvm::AMDGPU::DepCtr;
8265
8266 SMLoc DepCtrLoc = getLoc();
8267 StringRef DepCtrName = getTokenStr();
8268
8269 if (!skipToken(AsmToken::Identifier, "expected a counter name") ||
8270 !skipToken(AsmToken::LParen, "expected a left parenthesis"))
8271 return false;
8272
8273 int64_t ExprVal;
8274 if (!parseExpr(ExprVal))
8275 return false;
8276
8277 unsigned PrevOprMask = UsedOprMask;
8278 int CntVal = encodeDepCtr(DepCtrName, ExprVal, UsedOprMask, getSTI());
8279
8280 if (CntVal < 0) {
8281 depCtrError(DepCtrLoc, CntVal, DepCtrName);
8282 return false;
8283 }
8284
8285 if (!skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8286 return false;
8287
8288 if (trySkipToken(AsmToken::Amp) || trySkipToken(AsmToken::Comma)) {
8289 if (isToken(AsmToken::EndOfStatement)) {
8290 Error(getLoc(), "expected a counter name");
8291 return false;
8292 }
8293 }
8294
8295 int64_t CntValMask = PrevOprMask ^ UsedOprMask;
8296 DepCtr = (DepCtr & ~CntValMask) | CntVal;
8297 return true;
8298}
8299
8300ParseStatus AMDGPUAsmParser::parseDepCtr(OperandVector &Operands) {
8301 using namespace llvm::AMDGPU::DepCtr;
8302
8303 int64_t DepCtr = getDefaultDepCtrEncoding(getSTI());
8304 SMLoc Loc = getLoc();
8305
8306 if (isToken(AsmToken::Identifier) && peekToken().is(AsmToken::LParen)) {
8307 unsigned UsedOprMask = 0;
8308 while (!isToken(AsmToken::EndOfStatement)) {
8309 if (!parseDepCtr(DepCtr, UsedOprMask))
8310 return ParseStatus::Failure;
8311 }
8312 } else {
8313 if (!parseExpr(DepCtr))
8314 return ParseStatus::Failure;
8315 }
8316
8317 Operands.push_back(AMDGPUOperand::CreateImm(this, DepCtr, Loc));
8318 return ParseStatus::Success;
8319}
8320
8321bool AMDGPUOperand::isDepCtr() const { return isS16Imm(); }
8322
8323//===----------------------------------------------------------------------===//
8324// hwreg
8325//===----------------------------------------------------------------------===//
8326
8327ParseStatus AMDGPUAsmParser::parseHwregFunc(OperandInfoTy &HwReg,
8328 OperandInfoTy &Offset,
8329 OperandInfoTy &Width) {
8330 using namespace llvm::AMDGPU::Hwreg;
8331
8332 if (!trySkipId("hwreg", AsmToken::LParen))
8333 return ParseStatus::NoMatch;
8334
8335 // The register may be specified by name or using a numeric code
8336 HwReg.Loc = getLoc();
8337 if (isToken(AsmToken::Identifier) &&
8338 (HwReg.Val = getHwregId(getTokenStr(), getSTI())) != OPR_ID_UNKNOWN) {
8339 HwReg.IsSymbolic = true;
8340 lex(); // skip register name
8341 } else if (!parseExpr(HwReg.Val, "a register name")) {
8342 return ParseStatus::Failure;
8343 }
8344
8345 if (trySkipToken(AsmToken::RParen))
8346 return ParseStatus::Success;
8347
8348 // parse optional params
8349 if (!skipToken(AsmToken::Comma, "expected a comma or a closing parenthesis"))
8350 return ParseStatus::Failure;
8351
8352 Offset.Loc = getLoc();
8353 if (!parseExpr(Offset.Val))
8354 return ParseStatus::Failure;
8355
8356 if (!skipToken(AsmToken::Comma, "expected a comma"))
8357 return ParseStatus::Failure;
8358
8359 Width.Loc = getLoc();
8360 if (!parseExpr(Width.Val) ||
8361 !skipToken(AsmToken::RParen, "expected a closing parenthesis"))
8362 return ParseStatus::Failure;
8363
8364 return ParseStatus::Success;
8365}
8366
8367ParseStatus AMDGPUAsmParser::parseHwreg(OperandVector &Operands) {
8368 using namespace llvm::AMDGPU::Hwreg;
8369
8370 int64_t ImmVal = 0;
8371 SMLoc Loc = getLoc();
8372
8373 StructuredOpField HwReg("id", "hardware register", HwregId::Width,
8374 HwregId::Default);
8375 StructuredOpField Offset("offset", "bit offset", HwregOffset::Width,
8376 HwregOffset::Default);
8377 struct : StructuredOpField {
8378 using StructuredOpField::StructuredOpField;
8379 bool validate(AMDGPUAsmParser &Parser) const override {
8380 if (!isUIntN(Width, Val - 1))
8381 return Error(Parser, "only values from 1 to 32 are legal");
8382 return true;
8383 }
8384 } Width("size", "bitfield width", HwregSize::Width, HwregSize::Default);
8385 ParseStatus Res = parseStructuredOpFields({&HwReg, &Offset, &Width});
8386
8387 if (Res.isNoMatch())
8388 Res = parseHwregFunc(HwReg, Offset, Width);
8389
8390 if (Res.isSuccess()) {
8391 if (!validateStructuredOpFields({&HwReg, &Offset, &Width}))
8392 return ParseStatus::Failure;
8393 ImmVal = HwregEncoding::encode(HwReg.Val, Offset.Val, Width.Val);
8394 }
8395
8396 if (Res.isNoMatch() &&
8397 parseExpr(ImmVal, "a hwreg macro, structured immediate"))
8399
8400 if (!Res.isSuccess())
8401 return ParseStatus::Failure;
8402
8403 if (!isUInt<16>(ImmVal))
8404 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8405 Operands.push_back(
8406 AMDGPUOperand::CreateImm(this, ImmVal, Loc, AMDGPUOperand::ImmTyHwreg));
8407 return ParseStatus::Success;
8408}
8409
8410bool AMDGPUOperand::isHwreg() const { return isImmTy(ImmTyHwreg); }
8411
8412//===----------------------------------------------------------------------===//
8413// sendmsg
8414//===----------------------------------------------------------------------===//
8415
8416bool AMDGPUAsmParser::parseSendMsgBody(OperandInfoTy &Msg, OperandInfoTy &Op,
8417 OperandInfoTy &Stream) {
8418 using namespace llvm::AMDGPU::SendMsg;
8419
8420 Msg.Loc = getLoc();
8421 if (isToken(AsmToken::Identifier) &&
8422 (Msg.Val = getMsgId(getTokenStr(), getSTI())) != OPR_ID_UNKNOWN) {
8423 Msg.IsSymbolic = true;
8424 lex(); // skip message name
8425 } else if (!parseExpr(Msg.Val, "a message name")) {
8426 return false;
8427 }
8428
8429 if (trySkipToken(AsmToken::Comma)) {
8430 Op.IsDefined = true;
8431 Op.Loc = getLoc();
8432 if (isToken(AsmToken::Identifier) &&
8433 (Op.Val = getMsgOpId(Msg.Val, getTokenStr(), getSTI())) !=
8435 lex(); // skip operation name
8436 } else if (!parseExpr(Op.Val, "an operation name")) {
8437 return false;
8438 }
8439
8440 if (trySkipToken(AsmToken::Comma)) {
8441 Stream.IsDefined = true;
8442 Stream.Loc = getLoc();
8443 if (!parseExpr(Stream.Val))
8444 return false;
8445 }
8446 }
8447
8448 return skipToken(AsmToken::RParen, "expected a closing parenthesis");
8449}
8450
8451bool AMDGPUAsmParser::validateSendMsg(const OperandInfoTy &Msg,
8452 const OperandInfoTy &Op,
8453 const OperandInfoTy &Stream) {
8454 using namespace llvm::AMDGPU::SendMsg;
8455
8456 // Validation strictness depends on whether message is specified
8457 // in a symbolic or in a numeric form. In the latter case
8458 // only encoding possibility is checked.
8459 bool Strict = Msg.IsSymbolic;
8460
8461 if (Strict) {
8462 if (Msg.Val == OPR_ID_UNSUPPORTED) {
8463 Error(Msg.Loc, "specified message id is not supported on this GPU");
8464 return false;
8465 }
8466 } else {
8467 if (!isValidMsgId(Msg.Val, getSTI())) {
8468 Error(Msg.Loc, "invalid message id");
8469 return false;
8470 }
8471 }
8472 if (Strict && (msgRequiresOp(Msg.Val, getSTI()) != Op.IsDefined)) {
8473 if (Op.IsDefined) {
8474 Error(Op.Loc, "message does not support operations");
8475 } else {
8476 Error(Msg.Loc, "missing message operation");
8477 }
8478 return false;
8479 }
8480 if (!isValidMsgOp(Msg.Val, Op.Val, getSTI(), Strict)) {
8481 if (Op.Val == OPR_ID_UNSUPPORTED)
8482 Error(Op.Loc, "specified operation id is not supported on this GPU");
8483 else
8484 Error(Op.Loc, "invalid operation id");
8485 return false;
8486 }
8487 if (Strict && !msgSupportsStream(Msg.Val, Op.Val, getSTI()) &&
8488 Stream.IsDefined) {
8489 Error(Stream.Loc, "message operation does not support streams");
8490 return false;
8491 }
8492 if (!isValidMsgStream(Msg.Val, Op.Val, Stream.Val, getSTI(), Strict)) {
8493 Error(Stream.Loc, "invalid message stream id");
8494 return false;
8495 }
8496 return true;
8497}
8498
8499ParseStatus AMDGPUAsmParser::parseSendMsg(OperandVector &Operands) {
8500 using namespace llvm::AMDGPU::SendMsg;
8501
8502 int64_t ImmVal = 0;
8503 SMLoc Loc = getLoc();
8504
8505 if (trySkipId("sendmsg", AsmToken::LParen)) {
8506 OperandInfoTy Msg(OPR_ID_UNKNOWN);
8507 OperandInfoTy Op(OP_NONE_);
8508 OperandInfoTy Stream(STREAM_ID_NONE_);
8509 if (parseSendMsgBody(Msg, Op, Stream) && validateSendMsg(Msg, Op, Stream)) {
8510 ImmVal = encodeMsg(Msg.Val, Op.Val, Stream.Val);
8511 } else {
8512 return ParseStatus::Failure;
8513 }
8514 } else if (parseExpr(ImmVal, "a sendmsg macro")) {
8515 if (ImmVal < 0 || !isUInt<16>(ImmVal))
8516 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8517 } else {
8518 return ParseStatus::Failure;
8519 }
8520
8521 Operands.push_back(
8522 AMDGPUOperand::CreateImm(this, ImmVal, Loc, AMDGPUOperand::ImmTySendMsg));
8523 return ParseStatus::Success;
8524}
8525
8526bool AMDGPUOperand::isSendMsg() const { return isImmTy(ImmTySendMsg); }
8527
8528ParseStatus AMDGPUAsmParser::parseWaitEvent(OperandVector &Operands) {
8529 using namespace llvm::AMDGPU::WaitEvent;
8530
8531 SMLoc Loc = getLoc();
8532 int64_t ImmVal = 0;
8533
8534 StructuredOpField DontWaitExportReady("dont_wait_export_ready", "bit value",
8535 1, 0);
8536 StructuredOpField ExportReady("export_ready", "bit value", 1, 0);
8537
8538 StructuredOpField *TargetBitfield =
8539 isGFX11() ? &DontWaitExportReady : &ExportReady;
8540
8541 ParseStatus Res = parseStructuredOpFields({TargetBitfield});
8542 if (Res.isNoMatch() && parseExpr(ImmVal, "structured immediate"))
8544 else if (Res.isSuccess()) {
8545 if (!validateStructuredOpFields({TargetBitfield}))
8546 return ParseStatus::Failure;
8547 ImmVal = TargetBitfield->Val;
8548 }
8549
8550 if (!Res.isSuccess())
8551 return ParseStatus::Failure;
8552
8553 if (!isUInt<16>(ImmVal))
8554 return Error(Loc, "invalid immediate: only 16-bit values are legal");
8555
8556 Operands.push_back(AMDGPUOperand::CreateImm(this, ImmVal, Loc,
8557 AMDGPUOperand::ImmTyWaitEvent));
8558 return ParseStatus::Success;
8559}
8560
8561bool AMDGPUOperand::isWaitEvent() const { return isImmTy(ImmTyWaitEvent); }
8562
8563//===----------------------------------------------------------------------===//
8564// v_interp
8565//===----------------------------------------------------------------------===//
8566
8567ParseStatus AMDGPUAsmParser::parseInterpSlot(OperandVector &Operands) {
8568 StringRef Str;
8569 SMLoc S = getLoc();
8570
8571 if (!parseId(Str))
8572 return ParseStatus::NoMatch;
8573
8574 int Slot = StringSwitch<int>(Str)
8575 .Case("p10", 0)
8576 .Case("p20", 1)
8577 .Case("p0", 2)
8578 .Default(-1);
8579
8580 if (Slot == -1)
8581 return Error(S, "invalid interpolation slot");
8582
8583 Operands.push_back(
8584 AMDGPUOperand::CreateImm(this, Slot, S, AMDGPUOperand::ImmTyInterpSlot));
8585 return ParseStatus::Success;
8586}
8587
8588ParseStatus AMDGPUAsmParser::parseInterpAttr(OperandVector &Operands) {
8589 StringRef Str;
8590 SMLoc S = getLoc();
8591
8592 if (!parseId(Str))
8593 return ParseStatus::NoMatch;
8594
8595 if (!Str.starts_with("attr"))
8596 return Error(S, "invalid interpolation attribute");
8597
8598 StringRef Chan = Str.take_back(2);
8599 int AttrChan = StringSwitch<int>(Chan)
8600 .Case(".x", 0)
8601 .Case(".y", 1)
8602 .Case(".z", 2)
8603 .Case(".w", 3)
8604 .Default(-1);
8605 if (AttrChan == -1)
8606 return Error(S, "invalid or missing interpolation attribute channel");
8607
8608 Str = Str.drop_back(2).drop_front(4);
8609
8610 uint8_t Attr;
8611 if (Str.getAsInteger(10, Attr))
8612 return Error(S, "invalid or missing interpolation attribute number");
8613
8614 if (Attr > 32)
8615 return Error(S, "out of bounds interpolation attribute number");
8616
8617 SMLoc SChan = SMLoc::getFromPointer(Chan.data());
8618
8619 Operands.push_back(
8620 AMDGPUOperand::CreateImm(this, Attr, S, AMDGPUOperand::ImmTyInterpAttr));
8621 Operands.push_back(AMDGPUOperand::CreateImm(
8622 this, AttrChan, SChan, AMDGPUOperand::ImmTyInterpAttrChan));
8623 return ParseStatus::Success;
8624}
8625
8626//===----------------------------------------------------------------------===//
8627// exp
8628//===----------------------------------------------------------------------===//
8629
8630ParseStatus AMDGPUAsmParser::parseExpTgt(OperandVector &Operands) {
8631 using namespace llvm::AMDGPU::Exp;
8632
8633 StringRef Str;
8634 SMLoc S = getLoc();
8635
8636 if (!parseId(Str))
8637 return ParseStatus::NoMatch;
8638
8639 unsigned Id = getTgtId(Str);
8640 if (Id == ET_INVALID || !isSupportedTgtId(Id, getSTI()))
8641 return Error(S, (Id == ET_INVALID)
8642 ? "invalid exp target"
8643 : "exp target is not supported on this GPU");
8644
8645 Operands.push_back(
8646 AMDGPUOperand::CreateImm(this, Id, S, AMDGPUOperand::ImmTyExpTgt));
8647 return ParseStatus::Success;
8648}
8649
8650//===----------------------------------------------------------------------===//
8651// parser helpers
8652//===----------------------------------------------------------------------===//
8653
8654bool AMDGPUAsmParser::isId(const AsmToken &Token, const StringRef Id) const {
8655 return Token.is(AsmToken::Identifier) && Token.getString() == Id;
8656}
8657
8658bool AMDGPUAsmParser::isId(const StringRef Id) const {
8659 return isId(getToken(), Id);
8660}
8661
8662bool AMDGPUAsmParser::isToken(const AsmToken::TokenKind Kind) const {
8663 return getTokenKind() == Kind;
8664}
8665
8666StringRef AMDGPUAsmParser::getId() const {
8667 return isToken(AsmToken::Identifier) ? getTokenStr() : StringRef();
8668}
8669
8670bool AMDGPUAsmParser::trySkipId(const StringRef Id) {
8671 if (isId(Id)) {
8672 lex();
8673 return true;
8674 }
8675 return false;
8676}
8677
8678bool AMDGPUAsmParser::trySkipId(const StringRef Pref, const StringRef Id) {
8679 if (isToken(AsmToken::Identifier)) {
8680 StringRef Tok = getTokenStr();
8681 if (Tok.starts_with(Pref) && Tok.drop_front(Pref.size()) == Id) {
8682 lex();
8683 return true;
8684 }
8685 }
8686 return false;
8687}
8688
8689bool AMDGPUAsmParser::trySkipId(const StringRef Id,
8690 const AsmToken::TokenKind Kind) {
8691 if (isId(Id) && peekToken().is(Kind)) {
8692 lex();
8693 lex();
8694 return true;
8695 }
8696 return false;
8697}
8698
8699bool AMDGPUAsmParser::trySkipToken(const AsmToken::TokenKind Kind) {
8700 if (isToken(Kind)) {
8701 lex();
8702 return true;
8703 }
8704 return false;
8705}
8706
8707bool AMDGPUAsmParser::skipToken(const AsmToken::TokenKind Kind,
8708 const StringRef ErrMsg) {
8709 if (!trySkipToken(Kind)) {
8710 Error(getLoc(), ErrMsg);
8711 return false;
8712 }
8713 return true;
8714}
8715
8716bool AMDGPUAsmParser::parseExpr(int64_t &Imm, StringRef Expected) {
8717 SMLoc S = getLoc();
8718
8719 const MCExpr *Expr;
8720 if (Parser.parseExpression(Expr))
8721 return false;
8722
8723 if (Expr->evaluateAsAbsolute(Imm))
8724 return true;
8725
8726 if (Expected.empty()) {
8727 Error(S, "expected absolute expression");
8728 } else {
8729 Error(S,
8730 Twine("expected ", Expected) + Twine(" or an absolute expression"));
8731 }
8732 return false;
8733}
8734
8735bool AMDGPUAsmParser::parseExpr(OperandVector &Operands) {
8736 SMLoc S = getLoc();
8737
8738 const MCExpr *Expr;
8739 if (Parser.parseExpression(Expr))
8740 return false;
8741
8742 int64_t IntVal;
8743 if (Expr->evaluateAsAbsolute(IntVal)) {
8744 Operands.push_back(AMDGPUOperand::CreateImm(this, IntVal, S));
8745 } else {
8746 Operands.push_back(AMDGPUOperand::CreateExpr(this, Expr, S));
8747 }
8748 return true;
8749}
8750
8751bool AMDGPUAsmParser::parseString(StringRef &Val, const StringRef ErrMsg) {
8752 if (isToken(AsmToken::String)) {
8753 Val = getToken().getStringContents();
8754 lex();
8755 return true;
8756 }
8757 Error(getLoc(), ErrMsg);
8758 return false;
8759}
8760
8761bool AMDGPUAsmParser::parseId(StringRef &Val, const StringRef ErrMsg) {
8762 if (isToken(AsmToken::Identifier)) {
8763 Val = getTokenStr();
8764 lex();
8765 return true;
8766 }
8767 if (!ErrMsg.empty())
8768 Error(getLoc(), ErrMsg);
8769 return false;
8770}
8771
8772AsmToken AMDGPUAsmParser::getToken() const { return Parser.getTok(); }
8773
8774AsmToken AMDGPUAsmParser::peekToken(bool ShouldSkipSpace) {
8775 return isToken(AsmToken::EndOfStatement)
8776 ? getToken()
8777 : getLexer().peekTok(ShouldSkipSpace);
8778}
8779
8780void AMDGPUAsmParser::peekTokens(MutableArrayRef<AsmToken> Tokens) {
8781 auto TokCount = getLexer().peekTokens(Tokens);
8782
8783 for (auto Idx = TokCount; Idx < Tokens.size(); ++Idx)
8784 Tokens[Idx] = AsmToken(AsmToken::Error, "");
8785}
8786
8787AsmToken::TokenKind AMDGPUAsmParser::getTokenKind() const {
8788 return getLexer().getKind();
8789}
8790
8791SMLoc AMDGPUAsmParser::getLoc() const { return getToken().getLoc(); }
8792
8793StringRef AMDGPUAsmParser::getTokenStr() const {
8794 return getToken().getString();
8795}
8796
8797void AMDGPUAsmParser::lex() { Parser.Lex(); }
8798
8799const AMDGPUOperand &
8800AMDGPUAsmParser::findMCOperand(const OperandVector &Operands,
8801 int MCOpIdx) const {
8802 for (const auto &Op : Operands) {
8803 const AMDGPUOperand &TargetOp = static_cast<AMDGPUOperand &>(*Op);
8804 if (TargetOp.getMCOpIdx() == MCOpIdx)
8805 return TargetOp;
8806 }
8807 llvm_unreachable("no such MC operand!");
8808}
8809
8810SMLoc AMDGPUAsmParser::getInstLoc(const OperandVector &Operands) const {
8811 return ((AMDGPUOperand &)*Operands[0]).getStartLoc();
8812}
8813
8814// Returns one of the given locations that comes later in the source.
8815SMLoc AMDGPUAsmParser::getLaterLoc(SMLoc a, SMLoc b) {
8816 return a.getPointer() < b.getPointer() ? b : a;
8817}
8818
8819SMLoc AMDGPUAsmParser::getOperandLoc(const OperandVector &Operands,
8820 int MCOpIdx) const {
8821 return findMCOperand(Operands, MCOpIdx).getStartLoc();
8822}
8823
8824SMLoc AMDGPUAsmParser::getOperandLoc(
8825 std::function<bool(const AMDGPUOperand &)> Test,
8826 const OperandVector &Operands) const {
8827 for (unsigned i = Operands.size() - 1; i > 0; --i) {
8828 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
8829 if (Test(Op))
8830 return Op.getStartLoc();
8831 }
8832 return getInstLoc(Operands);
8833}
8834
8835SMLoc AMDGPUAsmParser::getImmLoc(AMDGPUOperand::ImmTy Type,
8836 const OperandVector &Operands) const {
8837 auto Test = [=](const AMDGPUOperand &Op) { return Op.isImmTy(Type); };
8838 return getOperandLoc(Test, Operands);
8839}
8840
8841ParseStatus
8842AMDGPUAsmParser::parseStructuredOpFields(ArrayRef<StructuredOpField *> Fields) {
8843 if (!trySkipToken(AsmToken::LCurly))
8844 return ParseStatus::NoMatch;
8845
8846 bool First = true;
8847 while (!trySkipToken(AsmToken::RCurly)) {
8848 if (!First &&
8849 !skipToken(AsmToken::Comma, "comma or closing brace expected"))
8850 return ParseStatus::Failure;
8851
8852 StringRef Id = getTokenStr();
8853 SMLoc IdLoc = getLoc();
8854 if (!skipToken(AsmToken::Identifier, "field name expected") ||
8855 !skipToken(AsmToken::Colon, "colon expected"))
8856 return ParseStatus::Failure;
8857
8858 const auto *I =
8859 find_if(Fields, [Id](StructuredOpField *F) { return F->Id == Id; });
8860 if (I == Fields.end())
8861 return Error(IdLoc, "unknown field");
8862 if ((*I)->IsDefined)
8863 return Error(IdLoc, "duplicate field");
8864
8865 // TODO: Support symbolic values.
8866 (*I)->Loc = getLoc();
8867 if (!parseExpr((*I)->Val))
8868 return ParseStatus::Failure;
8869 (*I)->IsDefined = true;
8870
8871 First = false;
8872 }
8873 return ParseStatus::Success;
8874}
8875
8876bool AMDGPUAsmParser::validateStructuredOpFields(
8878 return all_of(Fields, [this](const StructuredOpField *F) {
8879 return F->validate(*this);
8880 });
8881}
8882
8883//===----------------------------------------------------------------------===//
8884// swizzle
8885//===----------------------------------------------------------------------===//
8886
8888static unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask,
8889 const unsigned XorMask) {
8890 using namespace llvm::AMDGPU::Swizzle;
8891
8892 return BITMASK_PERM_ENC | (AndMask << BITMASK_AND_SHIFT) |
8893 (OrMask << BITMASK_OR_SHIFT) | (XorMask << BITMASK_XOR_SHIFT);
8894}
8895
8896bool AMDGPUAsmParser::parseSwizzleOperand(int64_t &Op, const unsigned MinVal,
8897 const unsigned MaxVal,
8898 const Twine &ErrMsg, SMLoc &Loc) {
8899 if (!skipToken(AsmToken::Comma, "expected a comma")) {
8900 return false;
8901 }
8902 Loc = getLoc();
8903 if (!parseExpr(Op)) {
8904 return false;
8905 }
8906 if (Op < MinVal || Op > MaxVal) {
8907 Error(Loc, ErrMsg);
8908 return false;
8909 }
8910
8911 return true;
8912}
8913
8914bool AMDGPUAsmParser::parseSwizzleOperands(const unsigned OpNum, int64_t *Op,
8915 const unsigned MinVal,
8916 const unsigned MaxVal,
8917 const StringRef ErrMsg) {
8918 SMLoc Loc;
8919 for (unsigned i = 0; i < OpNum; ++i) {
8920 if (!parseSwizzleOperand(Op[i], MinVal, MaxVal, ErrMsg, Loc))
8921 return false;
8922 }
8923
8924 return true;
8925}
8926
8927bool AMDGPUAsmParser::parseSwizzleQuadPerm(int64_t &Imm) {
8928 using namespace llvm::AMDGPU::Swizzle;
8929
8930 int64_t Lane[LANE_NUM];
8931 if (parseSwizzleOperands(LANE_NUM, Lane, 0, LANE_MAX,
8932 "expected a 2-bit lane id")) {
8934 for (unsigned I = 0; I < LANE_NUM; ++I) {
8935 Imm |= Lane[I] << (LANE_SHIFT * I);
8936 }
8937 return true;
8938 }
8939 return false;
8940}
8941
8942bool AMDGPUAsmParser::parseSwizzleBroadcast(int64_t &Imm) {
8943 using namespace llvm::AMDGPU::Swizzle;
8944
8945 SMLoc Loc;
8946 int64_t GroupSize;
8947 int64_t LaneIdx;
8948
8949 if (!parseSwizzleOperand(GroupSize, 2, 32,
8950 "group size must be in the interval [2,32]", Loc)) {
8951 return false;
8952 }
8953 if (!isPowerOf2_64(GroupSize)) {
8954 Error(Loc, "group size must be a power of two");
8955 return false;
8956 }
8957 if (parseSwizzleOperand(LaneIdx, 0, GroupSize - 1,
8958 "lane id must be in the interval [0,group size - 1]",
8959 Loc)) {
8960 Imm = encodeBitmaskPerm(BITMASK_MAX - GroupSize + 1, LaneIdx, 0);
8961 return true;
8962 }
8963 return false;
8964}
8965
8966bool AMDGPUAsmParser::parseSwizzleReverse(int64_t &Imm) {
8967 using namespace llvm::AMDGPU::Swizzle;
8968
8969 SMLoc Loc;
8970 int64_t GroupSize;
8971
8972 if (!parseSwizzleOperand(GroupSize, 2, 32,
8973 "group size must be in the interval [2,32]", Loc)) {
8974 return false;
8975 }
8976 if (!isPowerOf2_64(GroupSize)) {
8977 Error(Loc, "group size must be a power of two");
8978 return false;
8979 }
8980
8981 Imm = encodeBitmaskPerm(BITMASK_MAX, 0, GroupSize - 1);
8982 return true;
8983}
8984
8985bool AMDGPUAsmParser::parseSwizzleSwap(int64_t &Imm) {
8986 using namespace llvm::AMDGPU::Swizzle;
8987
8988 SMLoc Loc;
8989 int64_t GroupSize;
8990
8991 if (!parseSwizzleOperand(GroupSize, 1, 16,
8992 "group size must be in the interval [1,16]", Loc)) {
8993 return false;
8994 }
8995 if (!isPowerOf2_64(GroupSize)) {
8996 Error(Loc, "group size must be a power of two");
8997 return false;
8998 }
8999
9000 Imm = encodeBitmaskPerm(BITMASK_MAX, 0, GroupSize);
9001 return true;
9002}
9003
9004bool AMDGPUAsmParser::parseSwizzleBitmaskPerm(int64_t &Imm) {
9005 using namespace llvm::AMDGPU::Swizzle;
9006
9007 if (!skipToken(AsmToken::Comma, "expected a comma")) {
9008 return false;
9009 }
9010
9011 StringRef Ctl;
9012 SMLoc StrLoc = getLoc();
9013 if (!parseString(Ctl)) {
9014 return false;
9015 }
9016 if (Ctl.size() != BITMASK_WIDTH) {
9017 Error(StrLoc, "expected a 5-character mask");
9018 return false;
9019 }
9020
9021 unsigned AndMask = 0;
9022 unsigned OrMask = 0;
9023 unsigned XorMask = 0;
9024
9025 for (size_t i = 0; i < Ctl.size(); ++i) {
9026 unsigned Mask = 1 << (BITMASK_WIDTH - 1 - i);
9027 switch (Ctl[i]) {
9028 default:
9029 Error(StrLoc, "invalid mask");
9030 return false;
9031 case '0':
9032 break;
9033 case '1':
9034 OrMask |= Mask;
9035 break;
9036 case 'p':
9037 AndMask |= Mask;
9038 break;
9039 case 'i':
9040 AndMask |= Mask;
9041 XorMask |= Mask;
9042 break;
9043 }
9044 }
9045
9046 Imm = encodeBitmaskPerm(AndMask, OrMask, XorMask);
9047 return true;
9048}
9049
9050bool AMDGPUAsmParser::parseSwizzleFFT(int64_t &Imm) {
9051 using namespace llvm::AMDGPU::Swizzle;
9052
9053 if (!AMDGPU::isGFX9Plus(getSTI())) {
9054 Error(getLoc(), "FFT mode swizzle not supported on this GPU");
9055 return false;
9056 }
9057
9058 int64_t Swizzle;
9059 SMLoc Loc;
9060 if (!parseSwizzleOperand(Swizzle, 0, FFT_SWIZZLE_MAX,
9061 "FFT swizzle must be in the interval [0," +
9062 Twine(FFT_SWIZZLE_MAX) + Twine(']'),
9063 Loc))
9064 return false;
9065
9066 Imm = FFT_MODE_ENC | Swizzle;
9067 return true;
9068}
9069
9070bool AMDGPUAsmParser::parseSwizzleRotate(int64_t &Imm) {
9071 using namespace llvm::AMDGPU::Swizzle;
9072
9073 if (!AMDGPU::isGFX9Plus(getSTI())) {
9074 Error(getLoc(), "Rotate mode swizzle not supported on this GPU");
9075 return false;
9076 }
9077
9078 SMLoc Loc;
9079 int64_t Direction;
9080
9081 if (!parseSwizzleOperand(Direction, 0, 1,
9082 "direction must be 0 (left) or 1 (right)", Loc))
9083 return false;
9084
9085 int64_t RotateSize;
9086 if (!parseSwizzleOperand(
9087 RotateSize, 0, ROTATE_MAX_SIZE,
9088 "number of threads to rotate must be in the interval [0," +
9089 Twine(ROTATE_MAX_SIZE) + Twine(']'),
9090 Loc))
9091 return false;
9092
9094 (RotateSize << ROTATE_SIZE_SHIFT);
9095 return true;
9096}
9097
9098bool AMDGPUAsmParser::parseSwizzleOffset(int64_t &Imm) {
9099
9100 SMLoc OffsetLoc = getLoc();
9101
9102 if (!parseExpr(Imm, "a swizzle macro")) {
9103 return false;
9104 }
9105 if (!isUInt<16>(Imm)) {
9106 Error(OffsetLoc, "expected a 16-bit offset");
9107 return false;
9108 }
9109 return true;
9110}
9111
9112bool AMDGPUAsmParser::parseSwizzleMacro(int64_t &Imm) {
9113 using namespace llvm::AMDGPU::Swizzle;
9114
9115 if (skipToken(AsmToken::LParen, "expected a left parentheses")) {
9116
9117 SMLoc ModeLoc = getLoc();
9118 bool Ok = false;
9119
9120 if (trySkipId(IdSymbolic[ID_QUAD_PERM])) {
9121 Ok = parseSwizzleQuadPerm(Imm);
9122 } else if (trySkipId(IdSymbolic[ID_BITMASK_PERM])) {
9123 Ok = parseSwizzleBitmaskPerm(Imm);
9124 } else if (trySkipId(IdSymbolic[ID_BROADCAST])) {
9125 Ok = parseSwizzleBroadcast(Imm);
9126 } else if (trySkipId(IdSymbolic[ID_SWAP])) {
9127 Ok = parseSwizzleSwap(Imm);
9128 } else if (trySkipId(IdSymbolic[ID_REVERSE])) {
9129 Ok = parseSwizzleReverse(Imm);
9130 } else if (trySkipId(IdSymbolic[ID_FFT])) {
9131 Ok = parseSwizzleFFT(Imm);
9132 } else if (trySkipId(IdSymbolic[ID_ROTATE])) {
9133 Ok = parseSwizzleRotate(Imm);
9134 } else {
9135 Error(ModeLoc, "expected a swizzle mode");
9136 }
9137
9138 return Ok && skipToken(AsmToken::RParen, "expected a closing parentheses");
9139 }
9140
9141 return false;
9142}
9143
9144ParseStatus AMDGPUAsmParser::parseSwizzle(OperandVector &Operands) {
9145 SMLoc S = getLoc();
9146 int64_t Imm = 0;
9147
9148 if (trySkipId("offset")) {
9149
9150 bool Ok = false;
9151 if (skipToken(AsmToken::Colon, "expected a colon")) {
9152 if (trySkipId("swizzle")) {
9153 Ok = parseSwizzleMacro(Imm);
9154 } else {
9155 Ok = parseSwizzleOffset(Imm);
9156 }
9157 }
9158
9159 Operands.push_back(
9160 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTySwizzle));
9161
9163 }
9164 return ParseStatus::NoMatch;
9165}
9166
9167bool AMDGPUOperand::isSwizzle() const { return isImmTy(ImmTySwizzle); }
9168
9169//===----------------------------------------------------------------------===//
9170// VGPR Index Mode
9171//===----------------------------------------------------------------------===//
9172
9173int64_t AMDGPUAsmParser::parseGPRIdxMacro() {
9174
9175 using namespace llvm::AMDGPU::VGPRIndexMode;
9176
9177 if (trySkipToken(AsmToken::RParen)) {
9178 return OFF;
9179 }
9180
9181 int64_t Imm = 0;
9182
9183 while (true) {
9184 unsigned Mode = 0;
9185 SMLoc S = getLoc();
9186
9187 for (unsigned ModeId = ID_MIN; ModeId <= ID_MAX; ++ModeId) {
9188 if (trySkipId(IdSymbolic[ModeId])) {
9189 Mode = 1 << ModeId;
9190 break;
9191 }
9192 }
9193
9194 if (Mode == 0) {
9195 Error(S, (Imm == 0)
9196 ? "expected a VGPR index mode or a closing parenthesis"
9197 : "expected a VGPR index mode");
9198 return UNDEF;
9199 }
9200
9201 if (Imm & Mode) {
9202 Error(S, "duplicate VGPR index mode");
9203 return UNDEF;
9204 }
9205 Imm |= Mode;
9206
9207 if (trySkipToken(AsmToken::RParen))
9208 break;
9209 if (!skipToken(AsmToken::Comma,
9210 "expected a comma or a closing parenthesis"))
9211 return UNDEF;
9212 }
9213
9214 return Imm;
9215}
9216
9217ParseStatus AMDGPUAsmParser::parseGPRIdxMode(OperandVector &Operands) {
9218
9219 using namespace llvm::AMDGPU::VGPRIndexMode;
9220
9221 int64_t Imm = 0;
9222 SMLoc S = getLoc();
9223
9224 if (trySkipId("gpr_idx", AsmToken::LParen)) {
9225 Imm = parseGPRIdxMacro();
9226 if (Imm == UNDEF)
9227 return ParseStatus::Failure;
9228 } else {
9229 if (getParser().parseAbsoluteExpression(Imm))
9230 return ParseStatus::Failure;
9231 if (Imm < 0 || !isUInt<4>(Imm))
9232 return Error(S, "invalid immediate: only 4-bit values are legal");
9233 }
9234
9235 Operands.push_back(
9236 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTyGprIdxMode));
9237 return ParseStatus::Success;
9238}
9239
9240bool AMDGPUOperand::isGPRIdxMode() const { return isImmTy(ImmTyGprIdxMode); }
9241
9242//===----------------------------------------------------------------------===//
9243// sopp branch targets
9244//===----------------------------------------------------------------------===//
9245
9246ParseStatus AMDGPUAsmParser::parseSOPPBrTarget(OperandVector &Operands) {
9247
9248 // Make sure we are not parsing something
9249 // that looks like a label or an expression but is not.
9250 // This will improve error messages.
9251 if (isRegister() || isModifier())
9252 return ParseStatus::NoMatch;
9253
9254 if (!parseExpr(Operands))
9255 return ParseStatus::Failure;
9256
9257 AMDGPUOperand &Opr = ((AMDGPUOperand &)*Operands[Operands.size() - 1]);
9258 assert(Opr.isImm() || Opr.isExpr());
9259 SMLoc Loc = Opr.getStartLoc();
9260
9261 // Currently we do not support arbitrary expressions as branch targets.
9262 // Only labels and absolute expressions are accepted.
9263 if (Opr.isExpr() && !Opr.isSymbolRefExpr()) {
9264 Error(Loc, "expected an absolute expression or a label");
9265 } else if (Opr.isImm() && !Opr.isS16Imm()) {
9266 Error(Loc, "expected a 16-bit signed jump offset");
9267 }
9268
9269 return ParseStatus::Success;
9270}
9271
9272//===----------------------------------------------------------------------===//
9273// Boolean holding registers
9274//===----------------------------------------------------------------------===//
9275
9276ParseStatus AMDGPUAsmParser::parseBoolReg(OperandVector &Operands) {
9277 return parseReg(Operands);
9278}
9279
9280//===----------------------------------------------------------------------===//
9281// mubuf
9282//===----------------------------------------------------------------------===//
9283
9284void AMDGPUAsmParser::cvtMubufImpl(MCInst &Inst, const OperandVector &Operands,
9285 bool IsAtomic) {
9286 OptionalImmIndexMap OptionalIdx;
9287 unsigned FirstOperandIdx = 1;
9288 bool IsAtomicReturn = false;
9289
9290 if (IsAtomic) {
9291 IsAtomicReturn = SIInstrFlags::isAtomicRet(MII, Inst);
9292 }
9293
9294 for (unsigned i = FirstOperandIdx, e = Operands.size(); i != e; ++i) {
9295 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
9296
9297 // Add the register arguments
9298 if (Op.isReg()) {
9299 Op.addRegOperands(Inst, 1);
9300 // Insert a tied src for atomic return dst.
9301 // This cannot be postponed as subsequent calls to
9302 // addImmOperands rely on correct number of MC operands.
9303 if (IsAtomicReturn && i == FirstOperandIdx)
9304 Op.addRegOperands(Inst, 1);
9305 continue;
9306 }
9307
9308 // Handle the case where soffset is an immediate
9309 if (Op.isImm() && Op.getImmTy() == AMDGPUOperand::ImmTyNone) {
9310 Op.addImmOperands(Inst, 1);
9311 continue;
9312 }
9313
9314 // Handle tokens like 'offen' which are sometimes hard-coded into the
9315 // asm string. There are no MCInst operands for these.
9316 if (Op.isToken()) {
9317 continue;
9318 }
9319 assert(Op.isImm());
9320
9321 // Handle optional arguments
9322 OptionalIdx[Op.getImmTy()] = i;
9323 }
9324
9325 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9326 AMDGPUOperand::ImmTyOffset);
9327 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyCPol,
9328 0);
9329 // Parse a dummy operand as a placeholder for the SWZ operand. This enforces
9330 // agreement between MCInstrDesc.getNumOperands and MCInst.getNumOperands.
9332 // The LDS variants carry a trailing IsAsync operand. Parse a dummy the same
9333 // way as the SWZ operand.
9334 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::IsAsync))
9336}
9337
9338//===----------------------------------------------------------------------===//
9339// smrd
9340//===----------------------------------------------------------------------===//
9341
9342bool AMDGPUOperand::isSMRDOffset8() const {
9343 return isImmLiteral() && isUInt<8>(getImm());
9344}
9345
9346bool AMDGPUOperand::isSMEMOffset() const {
9347 // Offset range is checked later by validator.
9348 return isImmLiteral();
9349}
9350
9351bool AMDGPUOperand::isSMRDLiteralOffset() const {
9352 // 32-bit literals are only supported on CI and we only want to use them
9353 // when the offset is > 8-bits.
9354 return isImmLiteral() && !isUInt<8>(getImm()) && isUInt<32>(getImm());
9355}
9356
9357//===----------------------------------------------------------------------===//
9358// vop3
9359//===----------------------------------------------------------------------===//
9360
9361static bool ConvertOmodMul(int64_t &Mul) {
9362 if (Mul != 1 && Mul != 2 && Mul != 4)
9363 return false;
9364
9365 Mul >>= 1;
9366 return true;
9367}
9368
9369static bool ConvertOmodDiv(int64_t &Div) {
9370 if (Div == 1) {
9371 Div = 0;
9372 return true;
9373 }
9374
9375 if (Div == 2) {
9376 Div = 3;
9377 return true;
9378 }
9379
9380 return false;
9381}
9382
9383// For pre-gfx11 targets, both bound_ctrl:0 and bound_ctrl:1 are encoded as 1.
9384// This is intentional and ensures compatibility with sp3.
9385// See bug 35397 for details.
9386bool AMDGPUAsmParser::convertDppBoundCtrl(int64_t &BoundCtrl) {
9387 if (BoundCtrl == 0 || BoundCtrl == 1) {
9388 if (!isGFX11Plus())
9389 BoundCtrl = 1;
9390 return true;
9391 }
9392 return false;
9393}
9394
9395void AMDGPUAsmParser::onBeginOfFile() {
9396 if (!getParser().getStreamer().getTargetStreamer())
9397 return;
9398
9399 if (!getTargetStreamer().getTargetID())
9400 getTargetStreamer().initializeTargetID(getSTI(),
9401 /*ApplyFeatureString=*/true);
9402}
9403
9404void AMDGPUAsmParser::emitTargetDirective() {
9405 if (TargetDirectiveEmitted)
9406 return;
9407 TargetDirectiveEmitted = true;
9408
9409 if (!getParser().getStreamer().getTargetStreamer() ||
9410 getSTI().getTargetTriple().getArch() == Triple::r600)
9411 return;
9412
9413 if (isHsaAbi(getSTI()))
9414 getTargetStreamer().EmitDirectiveAMDGCNTarget();
9415}
9416
9417/// Parse AMDGPU specific expressions.
9418///
9419/// expr ::= or(expr, ...) |
9420/// max(expr, ...) |
9421/// min(expr, ...)
9422///
9423bool AMDGPUAsmParser::parsePrimaryExpr(const MCExpr *&Res, SMLoc &EndLoc) {
9424 using AGVK = AMDGPUMCExpr::VariantKind;
9425
9426 if (isToken(AsmToken::Identifier)) {
9427 StringRef TokenId = getTokenStr();
9428 AGVK VK = StringSwitch<AGVK>(TokenId)
9429 .Case("max", AGVK::AGVK_Max)
9430 .Case("min", AGVK::AGVK_Min)
9431 .Case("or", AGVK::AGVK_Or)
9432 .Case("extrasgprs", AGVK::AGVK_ExtraSGPRs)
9433 .Case("totalnumvgprs", AGVK::AGVK_TotalNumVGPRs)
9434 .Case("alignto", AGVK::AGVK_AlignTo)
9435 .Case("occupancy", AGVK::AGVK_Occupancy)
9436 .Case("instprefsize", AGVK::AGVK_InstPrefSize)
9437 .Default(AGVK::AGVK_None);
9438
9439 if (VK != AGVK::AGVK_None && peekToken().is(AsmToken::LParen)) {
9441 uint64_t CommaCount = 0;
9442 lex(); // Eat Arg ('or', 'max', 'occupancy', etc.)
9443 lex(); // Eat '('
9444 while (true) {
9445 if (trySkipToken(AsmToken::RParen)) {
9446 if (Exprs.empty()) {
9447 Error(getToken().getLoc(),
9448 "empty " + Twine(TokenId) + " expression");
9449 return true;
9450 }
9451 if (CommaCount + 1 != Exprs.size()) {
9452 Error(getToken().getLoc(),
9453 "mismatch of commas in " + Twine(TokenId) + " expression");
9454 return true;
9455 }
9456 if (unsigned Expected = AMDGPUMCExpr::getNumExpectedArgs(VK);
9457 Expected && Exprs.size() != Expected) {
9458 Error(getToken().getLoc(), Twine(TokenId) + " expression expects " +
9459 Twine(Expected) + " operands");
9460 return true;
9461 }
9462 Res = AMDGPUMCExpr::create(VK, Exprs, getContext());
9463 return false;
9464 }
9465 const MCExpr *Expr;
9466 if (getParser().parseExpression(Expr, EndLoc))
9467 return true;
9468 Exprs.push_back(Expr);
9469 bool LastTokenWasComma = trySkipToken(AsmToken::Comma);
9470 if (LastTokenWasComma)
9471 CommaCount++;
9472 if (!LastTokenWasComma && !isToken(AsmToken::RParen)) {
9473 Error(getToken().getLoc(),
9474 "unexpected token in " + Twine(TokenId) + " expression");
9475 return true;
9476 }
9477 }
9478 }
9479 }
9480 return getParser().parsePrimaryExpr(Res, EndLoc, nullptr);
9481}
9482
9483ParseStatus AMDGPUAsmParser::parseOModSI(OperandVector &Operands) {
9484 StringRef Name = getTokenStr();
9485 if (Name == "mul") {
9486 return parseIntWithPrefix("mul", Operands, AMDGPUOperand::ImmTyOModSI,
9488 }
9489
9490 if (Name == "div") {
9491 return parseIntWithPrefix("div", Operands, AMDGPUOperand::ImmTyOModSI,
9493 }
9494
9495 return ParseStatus::NoMatch;
9496}
9497
9498// Determines which bit DST_OP_SEL occupies in the op_sel operand according to
9499// the number of src operands present, then copies that bit into src0_modifiers.
9500static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI) {
9501 int Opc = Inst.getOpcode();
9502 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9503 if (OpSelIdx == -1)
9504 return;
9505
9506 int SrcNum;
9507 const AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9508 AMDGPU::OpName::src2};
9509 for (SrcNum = 0; SrcNum < 3 && AMDGPU::hasNamedOperand(Opc, Ops[SrcNum]);
9510 ++SrcNum)
9511 ;
9512 assert(SrcNum > 0);
9513
9514 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9515
9516 int DstIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst);
9517 if (DstIdx == -1)
9518 return;
9519
9520 const MCOperand &DstOp = Inst.getOperand(DstIdx);
9521 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src0_modifiers);
9522 uint32_t ModVal = Inst.getOperand(ModIdx).getImm();
9523 if (DstOp.isReg() &&
9524 MRI.getRegClass(AMDGPU::VGPR_16RegClassID).contains(DstOp.getReg())) {
9525 if (AMDGPU::isHi16Reg(DstOp.getReg(), MRI))
9526 ModVal |= SISrcMods::DST_OP_SEL;
9527 } else {
9528 if ((OpSel & (1 << SrcNum)) != 0)
9529 ModVal |= SISrcMods::DST_OP_SEL;
9530 }
9531 Inst.getOperand(ModIdx).setImm(ModVal);
9532}
9533
9534void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst,
9535 const OperandVector &Operands) {
9536 cvtVOP3P(Inst, Operands);
9537 cvtVOP3DstOpSelOnly(Inst, *getMRI());
9538}
9539
9540void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst, const OperandVector &Operands,
9541 OptionalImmIndexMap &OptionalIdx) {
9542 cvtVOP3P(Inst, Operands, OptionalIdx);
9543 cvtVOP3DstOpSelOnly(Inst, *getMRI());
9544}
9545
9546static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum) {
9547 return
9548 // 1. This operand is input modifiers
9549 Desc.operands()[OpNum].OperandType == AMDGPU::OPERAND_INPUT_MODS
9550 // 2. This is not last operand
9551 && Desc.NumOperands > (OpNum + 1)
9552 // 3. Next operand is register class
9553 && Desc.operands()[OpNum + 1].RegClass != -1
9554 // 4. Next register is not tied to any other operand
9555 && Desc.getOperandConstraint(OpNum + 1,
9557}
9558
9559void AMDGPUAsmParser::cvtOpSelHelper(MCInst &Inst, unsigned OpSel) {
9560 unsigned Opc = Inst.getOpcode();
9561 constexpr AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9562 AMDGPU::OpName::src2};
9563 constexpr AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9564 AMDGPU::OpName::src1_modifiers,
9565 AMDGPU::OpName::src2_modifiers};
9566 for (int J = 0; J < 3; ++J) {
9567 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, Ops[J]);
9568 if (OpIdx == -1)
9569 // Some instructions, e.g. v_interp_p2_f16 in GFX9, have src0, src2, but
9570 // no src1. So continue instead of break.
9571 continue;
9572
9573 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
9574 uint32_t ModVal = Inst.getOperand(ModIdx).getImm();
9575
9576 if ((OpSel & (1 << J)) != 0)
9577 ModVal |= SISrcMods::OP_SEL_0;
9578 // op_sel[3] is encoded in src0_modifiers.
9579 if (ModOps[J] == AMDGPU::OpName::src0_modifiers && (OpSel & (1 << 3)) != 0)
9580 ModVal |= SISrcMods::DST_OP_SEL;
9581
9582 Inst.getOperand(ModIdx).setImm(ModVal);
9583 }
9584}
9585
9586void AMDGPUAsmParser::cvtVOP3Interp(MCInst &Inst,
9587 const OperandVector &Operands) {
9588 OptionalImmIndexMap OptionalIdx;
9589 unsigned Opc = Inst.getOpcode();
9590
9591 unsigned I = 1;
9592 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9593 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9594 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9595 }
9596
9597 for (unsigned E = Operands.size(); I != E; ++I) {
9598 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9600 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9601 } else if (Op.isInterpSlot() || Op.isInterpAttr() ||
9602 Op.isInterpAttrChan()) {
9603 Inst.addOperand(MCOperand::createImm(Op.getImm()));
9604 } else if (Op.isImmModifier()) {
9605 OptionalIdx[Op.getImmTy()] = I;
9606 } else {
9607 llvm_unreachable("unhandled operand type");
9608 }
9609 }
9610
9611 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::high))
9612 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9613 AMDGPUOperand::ImmTyHigh);
9614
9615 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
9616 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9617 AMDGPUOperand::ImmTyClamp);
9618
9619 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
9620 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9621 AMDGPUOperand::ImmTyOModSI);
9622
9623 // Some v_interp instructions use op_sel[3] for dst.
9624 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
9625 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9626 AMDGPUOperand::ImmTyOpSel);
9627 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9628 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9629
9630 cvtOpSelHelper(Inst, OpSel);
9631 }
9632}
9633
9634void AMDGPUAsmParser::cvtVINTERP(MCInst &Inst, const OperandVector &Operands) {
9635 OptionalImmIndexMap OptionalIdx;
9636 unsigned Opc = Inst.getOpcode();
9637
9638 unsigned I = 1;
9639 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9640 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9641 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9642 }
9643
9644 for (unsigned E = Operands.size(); I != E; ++I) {
9645 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9647 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9648 } else if (Op.isImmModifier()) {
9649 OptionalIdx[Op.getImmTy()] = I;
9650 } else {
9651 llvm_unreachable("unhandled operand type");
9652 }
9653 }
9654
9655 addOptionalImmOperand(Inst, Operands, OptionalIdx, AMDGPUOperand::ImmTyClamp);
9656
9657 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9658 if (OpSelIdx != -1)
9659 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9660 AMDGPUOperand::ImmTyOpSel);
9661
9662 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9663 AMDGPUOperand::ImmTyWaitEXP);
9664
9665 if (OpSelIdx == -1)
9666 return;
9667
9668 unsigned OpSel = Inst.getOperand(OpSelIdx).getImm();
9669 cvtOpSelHelper(Inst, OpSel);
9670}
9671
9672void AMDGPUAsmParser::cvtScaledMFMA(MCInst &Inst,
9673 const OperandVector &Operands) {
9674 OptionalImmIndexMap OptionalIdx;
9675 unsigned Opc = Inst.getOpcode();
9676 unsigned I = 1;
9677 int CbszOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::cbsz);
9678
9679 const MCInstrDesc &Desc = MII.get(Opc);
9680
9681 for (unsigned J = 0; J < Desc.getNumDefs(); ++J)
9682 static_cast<AMDGPUOperand &>(*Operands[I++]).addRegOperands(Inst, 1);
9683
9684 for (unsigned E = Operands.size(); I != E; ++I) {
9685 AMDGPUOperand &Op = static_cast<AMDGPUOperand &>(*Operands[I]);
9686 int NumOperands = Inst.getNumOperands();
9687 // The order of operands in MCInst and parsed operands are different.
9688 // Adding dummy cbsz and blgp operands at corresponding MCInst operand
9689 // indices for parsing scale values correctly.
9690 if (NumOperands == CbszOpIdx) {
9693 }
9694 if (isRegOrImmWithInputMods(Desc, NumOperands)) {
9695 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9696 } else if (Op.isImmModifier()) {
9697 OptionalIdx[Op.getImmTy()] = I;
9698 } else {
9699 Op.addRegOrImmOperands(Inst, 1);
9700 }
9701 }
9702
9703 // Insert CBSZ and BLGP operands for F8F6F4 variants
9704 auto CbszIdx = OptionalIdx.find(AMDGPUOperand::ImmTyCBSZ);
9705 if (CbszIdx != OptionalIdx.end()) {
9706 int CbszVal = ((AMDGPUOperand &)*Operands[CbszIdx->second]).getImm();
9707 Inst.getOperand(CbszOpIdx).setImm(CbszVal);
9708 }
9709
9710 int BlgpOpIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::blgp);
9711 auto BlgpIdx = OptionalIdx.find(AMDGPUOperand::ImmTyBLGP);
9712 if (BlgpIdx != OptionalIdx.end()) {
9713 int BlgpVal = ((AMDGPUOperand &)*Operands[BlgpIdx->second]).getImm();
9714 Inst.getOperand(BlgpOpIdx).setImm(BlgpVal);
9715 }
9716
9717 // Add dummy src_modifiers
9720
9721 // Handle op_sel fields
9722
9723 unsigned OpSel = 0;
9724 auto OpselIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSel);
9725 if (OpselIdx != OptionalIdx.end()) {
9726 OpSel = static_cast<const AMDGPUOperand &>(*Operands[OpselIdx->second])
9727 .getImm();
9728 }
9729
9730 unsigned OpSelHi = 0;
9731 auto OpselHiIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSelHi);
9732 if (OpselHiIdx != OptionalIdx.end()) {
9733 OpSelHi = static_cast<const AMDGPUOperand &>(*Operands[OpselHiIdx->second])
9734 .getImm();
9735 }
9736 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9737 AMDGPU::OpName::src1_modifiers};
9738
9739 for (unsigned J = 0; J < 2; ++J) {
9740 unsigned ModVal = 0;
9741 if (OpSel & (1 << J))
9742 ModVal |= SISrcMods::OP_SEL_0;
9743 if (OpSelHi & (1 << J))
9744 ModVal |= SISrcMods::OP_SEL_1;
9745
9746 const int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
9747 Inst.getOperand(ModIdx).setImm(ModVal);
9748 }
9749}
9750
9751void AMDGPUAsmParser::cvtVOP3(MCInst &Inst, const OperandVector &Operands,
9752 OptionalImmIndexMap &OptionalIdx) {
9753 unsigned Opc = Inst.getOpcode();
9754
9755 unsigned I = 1;
9756 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
9757 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
9758 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
9759 }
9760
9761 for (unsigned E = Operands.size(); I != E; ++I) {
9762 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
9764 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9765 } else if (Op.isImmModifier()) {
9766 OptionalIdx[Op.getImmTy()] = I;
9767 } else {
9768 Op.addRegOrImmOperands(Inst, 1);
9769 }
9770 }
9771
9772 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::scale_sel))
9773 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9774 AMDGPUOperand::ImmTyScaleSel);
9775
9776 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
9777 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9778 AMDGPUOperand::ImmTyClamp);
9779
9780 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::byte_sel)) {
9781 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::vdst_in))
9782 Inst.addOperand(Inst.getOperand(0));
9783 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9784 AMDGPUOperand::ImmTyByteSel);
9785 }
9786
9787 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
9788 addOptionalImmOperand(Inst, Operands, OptionalIdx,
9789 AMDGPUOperand::ImmTyOModSI);
9790
9791 // Special case v_mac_{f16, f32} and v_fmac_{f16, f32} (gfx906/gfx10+):
9792 // it has src2 register operand that is tied to dst operand
9793 // we don't allow modifiers for this operand in assembler so src2_modifiers
9794 // should be 0.
9795 if (isMAC(Opc)) {
9796 auto *it = Inst.begin();
9797 std::advance(
9798 it, AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2_modifiers));
9799 it = Inst.insert(it, MCOperand::createImm(0)); // no modifiers for src2
9800 ++it;
9801 // Copy the operand to ensure it's not invalidated when Inst grows.
9802 Inst.insert(it, MCOperand(Inst.getOperand(0))); // src2 = dst
9803 }
9804}
9805
9806void AMDGPUAsmParser::cvtVOP3(MCInst &Inst, const OperandVector &Operands) {
9807 OptionalImmIndexMap OptionalIdx;
9808 cvtVOP3(Inst, Operands, OptionalIdx);
9809}
9810
9811void AMDGPUAsmParser::cvtVOP3P(MCInst &Inst, const OperandVector &Operands,
9812 OptionalImmIndexMap &OptIdx) {
9813 const int Opc = Inst.getOpcode();
9814
9815 const bool IsPacked = SIInstrFlags::isPacked(MII, Inst);
9816
9817 if (Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_F16_vi ||
9818 Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_BF16_vi ||
9819 Opc == AMDGPU::V_CVT_SR_BF8_F32_vi ||
9820 Opc == AMDGPU::V_CVT_SR_FP8_F32_vi ||
9821 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx11 ||
9822 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx11 ||
9823 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx12 ||
9824 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx12 ||
9825 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx13 ||
9826 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx13) {
9827 Inst.addOperand(MCOperand::createImm(0)); // Placeholder for src2_mods
9828 Inst.addOperand(Inst.getOperand(0));
9829 }
9830
9831 // Append vdst_in only if a previous converter (cvtVOP3DPP for DPP variants,
9832 // cvtVOP3 for byte_sel variants) hasn't already placed it. Use the position
9833 // of the named operand to detect that, the same way cvtVOP3DPP does
9834 // internally.
9835 int VdstInIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
9836 if (VdstInIdx != -1 && VdstInIdx == static_cast<int>(Inst.getNumOperands()))
9837 Inst.addOperand(Inst.getOperand(0));
9838
9839 int BitOp3Idx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::bitop3);
9840 if (BitOp3Idx != -1) {
9841 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyBitOp3);
9842 }
9843
9844 // FIXME: This is messy. Parse the modifiers as if it was a normal VOP3
9845 // instruction, and then figure out where to actually put the modifiers
9846
9847 int OpSelIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel);
9848 if (OpSelIdx != -1) {
9849 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyOpSel);
9850 }
9851
9852 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel_hi);
9853 if (OpSelHiIdx != -1) {
9854 int DefaultVal = IsPacked ? -1 : 0;
9855 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyOpSelHi,
9856 DefaultVal);
9857 }
9858
9859 int MatrixAFMTIdx =
9860 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_fmt);
9861 if (MatrixAFMTIdx != -1) {
9862 addOptionalImmOperand(Inst, Operands, OptIdx,
9863 AMDGPUOperand::ImmTyMatrixAFMT, 0);
9864 }
9865
9866 int MatrixBFMTIdx =
9867 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_fmt);
9868 if (MatrixBFMTIdx != -1) {
9869 addOptionalImmOperand(Inst, Operands, OptIdx,
9870 AMDGPUOperand::ImmTyMatrixBFMT, 0);
9871 }
9872
9873 int MatrixAScaleIdx =
9874 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale);
9875 if (MatrixAScaleIdx != -1) {
9876 addOptionalImmOperand(Inst, Operands, OptIdx,
9877 AMDGPUOperand::ImmTyMatrixAScale, 0);
9878 }
9879
9880 int MatrixBScaleIdx =
9881 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale);
9882 if (MatrixBScaleIdx != -1) {
9883 addOptionalImmOperand(Inst, Operands, OptIdx,
9884 AMDGPUOperand::ImmTyMatrixBScale, 0);
9885 }
9886
9887 int MatrixAScaleFmtIdx =
9888 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_a_scale_fmt);
9889 if (MatrixAScaleFmtIdx != -1) {
9890 addOptionalImmOperand(Inst, Operands, OptIdx,
9891 AMDGPUOperand::ImmTyMatrixAScaleFmt, 0);
9892 }
9893
9894 int MatrixBScaleFmtIdx =
9895 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::matrix_b_scale_fmt);
9896 if (MatrixBScaleFmtIdx != -1) {
9897 addOptionalImmOperand(Inst, Operands, OptIdx,
9898 AMDGPUOperand::ImmTyMatrixBScaleFmt, 0);
9899 }
9900
9901 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::matrix_a_reuse))
9902 addOptionalImmOperand(Inst, Operands, OptIdx,
9903 AMDGPUOperand::ImmTyMatrixAReuse, 0);
9904
9905 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::matrix_b_reuse))
9906 addOptionalImmOperand(Inst, Operands, OptIdx,
9907 AMDGPUOperand::ImmTyMatrixBReuse, 0);
9908
9909 int NegLoIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::neg_lo);
9910 if (NegLoIdx != -1)
9911 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyNegLo);
9912
9913 int NegHiIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::neg_hi);
9914 if (NegHiIdx != -1)
9915 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyNegHi);
9916
9917 const AMDGPU::OpName Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9918 AMDGPU::OpName::src2};
9919 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9920 AMDGPU::OpName::src1_modifiers,
9921 AMDGPU::OpName::src2_modifiers};
9922
9923 unsigned OpSel = 0;
9924 unsigned OpSelHi = 0;
9925 unsigned NegLo = 0;
9926 unsigned NegHi = 0;
9927
9928 if (OpSelIdx != -1)
9929 OpSel = Inst.getOperand(OpSelIdx).getImm();
9930
9931 if (OpSelHiIdx != -1)
9932 OpSelHi = Inst.getOperand(OpSelHiIdx).getImm();
9933
9934 if (NegLoIdx != -1)
9935 NegLo = Inst.getOperand(NegLoIdx).getImm();
9936
9937 if (NegHiIdx != -1)
9938 NegHi = Inst.getOperand(NegHiIdx).getImm();
9939
9940 for (int J = 0; J < 3; ++J) {
9941 int OpIdx = AMDGPU::getNamedOperandIdx(Opc, Ops[J]);
9942 if (OpIdx == -1)
9943 break;
9944
9945 int ModIdx = AMDGPU::getNamedOperandIdx(Opc, ModOps[J]);
9946
9947 if (ModIdx == -1)
9948 continue;
9949
9950 // For MAC instructions, src2 is tied to vdst and its op_sel bit
9951 // is not encoded.
9952 if (AMDGPU::isMAC(Opc) && ModOps[J] == AMDGPU::OpName::src2_modifiers)
9953 continue;
9954
9955 uint32_t ModVal = 0;
9956
9957 const MCOperand &SrcOp = Inst.getOperand(OpIdx);
9958 if (SrcOp.isReg() && getMRI()
9959 ->getRegClass(AMDGPU::VGPR_16RegClassID)
9960 .contains(SrcOp.getReg())) {
9961 bool VGPRSuffixIsHi = AMDGPU::isHi16Reg(SrcOp.getReg(), *getMRI());
9962 if (VGPRSuffixIsHi)
9963 ModVal |= SISrcMods::OP_SEL_0;
9964 } else {
9965 if ((OpSel & (1 << J)) != 0)
9966 ModVal |= SISrcMods::OP_SEL_0;
9967 }
9968
9969 if ((OpSelHi & (1 << J)) != 0)
9970 ModVal |= SISrcMods::OP_SEL_1;
9971
9972 if ((NegLo & (1 << J)) != 0)
9973 ModVal |= SISrcMods::NEG;
9974
9975 if ((NegHi & (1 << J)) != 0)
9976 ModVal |= SISrcMods::NEG_HI;
9977
9978 Inst.getOperand(ModIdx).setImm(Inst.getOperand(ModIdx).getImm() | ModVal);
9979 }
9980}
9981
9982void AMDGPUAsmParser::cvtVOP3P(MCInst &Inst, const OperandVector &Operands) {
9983 OptionalImmIndexMap OptIdx;
9984 cvtVOP3(Inst, Operands, OptIdx);
9985 cvtVOP3P(Inst, Operands, OptIdx);
9986}
9987
9989 unsigned i, unsigned Opc,
9990 AMDGPU::OpName OpName) {
9991 if (AMDGPU::getNamedOperandIdx(Opc, OpName) != -1)
9992 ((AMDGPUOperand &)*Operands[i]).addRegOrImmWithFPInputModsOperands(Inst, 2);
9993 else
9994 ((AMDGPUOperand &)*Operands[i]).addRegOperands(Inst, 1);
9995}
9996
9997void AMDGPUAsmParser::cvtSWMMAC(MCInst &Inst, const OperandVector &Operands) {
9998 unsigned Opc = Inst.getOpcode();
9999
10000 ((AMDGPUOperand &)*Operands[1]).addRegOperands(Inst, 1);
10001 addSrcModifiersAndSrc(Inst, Operands, 2, Opc, AMDGPU::OpName::src0_modifiers);
10002 addSrcModifiersAndSrc(Inst, Operands, 3, Opc, AMDGPU::OpName::src1_modifiers);
10003 ((AMDGPUOperand &)*Operands[1]).addRegOperands(Inst, 1); // srcTiedDef
10004 ((AMDGPUOperand &)*Operands[4]).addRegOperands(Inst, 1); // src2
10005
10006 OptionalImmIndexMap OptIdx;
10007 for (unsigned i = 5; i < Operands.size(); ++i) {
10008 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[i]);
10009 OptIdx[Op.getImmTy()] = i;
10010 }
10011
10012 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_8bit))
10013 addOptionalImmOperand(Inst, Operands, OptIdx,
10014 AMDGPUOperand::ImmTyIndexKey8bit);
10015
10016 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_16bit))
10017 addOptionalImmOperand(Inst, Operands, OptIdx,
10018 AMDGPUOperand::ImmTyIndexKey16bit);
10019
10020 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::index_key_32bit))
10021 addOptionalImmOperand(Inst, Operands, OptIdx,
10022 AMDGPUOperand::ImmTyIndexKey32bit);
10023
10024 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
10025 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyClamp);
10026
10027 cvtVOP3P(Inst, Operands, OptIdx);
10028}
10029
10030//===----------------------------------------------------------------------===//
10031// VOPD
10032//===----------------------------------------------------------------------===//
10033
10034ParseStatus AMDGPUAsmParser::parseVOPD(OperandVector &Operands) {
10035 if (!hasVOPD(getSTI()))
10036 return ParseStatus::NoMatch;
10037
10038 if (isToken(AsmToken::Colon) && peekToken(false).is(AsmToken::Colon)) {
10039 SMLoc S = getLoc();
10040 lex();
10041 lex();
10042 Operands.push_back(AMDGPUOperand::CreateToken(this, "::", S));
10043 SMLoc OpYLoc = getLoc();
10044 StringRef OpYName;
10045 if (isToken(AsmToken::Identifier) && !Parser.parseIdentifier(OpYName)) {
10046 Operands.push_back(AMDGPUOperand::CreateToken(this, OpYName, OpYLoc));
10047 return ParseStatus::Success;
10048 }
10049 return Error(OpYLoc, "expected a VOPDY instruction after ::");
10050 }
10051 return ParseStatus::NoMatch;
10052}
10053
10054// Create VOPD MCInst operands using parsed assembler operands.
10055void AMDGPUAsmParser::cvtVOPD(MCInst &Inst, const OperandVector &Operands) {
10056 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10057
10058 auto addOp = [&](uint16_t ParsedOprIdx) { // NOLINT:function pointer
10059 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[ParsedOprIdx]);
10061 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10062 return;
10063 }
10064 if (Op.isReg()) {
10065 Op.addRegOperands(Inst, 1);
10066 return;
10067 }
10068 if (Op.isImm() || Op.isExpr()) {
10069 Op.addImmOperands(Inst, 1);
10070 return;
10071 }
10072 llvm_unreachable("Unhandled operand type in cvtVOPD");
10073 };
10074
10075 const auto &InstInfo = getVOPDInstInfo(Inst.getOpcode(), &MII);
10076
10077 // MCInst operands are ordered as follows:
10078 // dstX, dstY, src0X [, other OpX operands], src0Y [, other OpY operands]
10079
10080 for (auto CompIdx : VOPD::COMPONENTS) {
10081 addOp(InstInfo[CompIdx].getIndexOfDstInParsedOperands());
10082 }
10083
10084 for (auto CompIdx : VOPD::COMPONENTS) {
10085 const auto &CInfo = InstInfo[CompIdx];
10086 auto CompSrcOperandsNum = InstInfo[CompIdx].getCompParsedSrcOperandsNum();
10087 for (unsigned CompSrcIdx = 0; CompSrcIdx < CompSrcOperandsNum; ++CompSrcIdx)
10088 addOp(CInfo.getIndexOfSrcInParsedOperands(CompSrcIdx));
10089 if (CInfo.hasSrc2Acc())
10090 addOp(CInfo.getIndexOfDstInParsedOperands());
10091 }
10092
10093 int BitOp3Idx =
10094 AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::bitop3);
10095 if (BitOp3Idx != -1) {
10096 OptionalImmIndexMap OptIdx;
10097 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands.back());
10098 if (Op.isImm())
10099 OptIdx[Op.getImmTy()] = Operands.size() - 1;
10100
10101 addOptionalImmOperand(Inst, Operands, OptIdx, AMDGPUOperand::ImmTyBitOp3);
10102 }
10103}
10104
10105//===----------------------------------------------------------------------===//
10106// dpp
10107//===----------------------------------------------------------------------===//
10108
10109bool AMDGPUOperand::isDPP8() const { return isImmTy(ImmTyDPP8); }
10110
10111bool AMDGPUOperand::isDPPCtrl() const {
10112 using namespace AMDGPU::DPP;
10113
10114 bool result = isImm() && getImmTy() == ImmTyDppCtrl && isUInt<9>(getImm());
10115 if (result) {
10116 int64_t Imm = getImm();
10117 return (Imm >= DppCtrl::QUAD_PERM_FIRST &&
10118 Imm <= DppCtrl::QUAD_PERM_LAST) ||
10119 (Imm >= DppCtrl::ROW_SHL_FIRST && Imm <= DppCtrl::ROW_SHL_LAST) ||
10120 (Imm >= DppCtrl::ROW_SHR_FIRST && Imm <= DppCtrl::ROW_SHR_LAST) ||
10121 (Imm >= DppCtrl::ROW_ROR_FIRST && Imm <= DppCtrl::ROW_ROR_LAST) ||
10122 (Imm == DppCtrl::WAVE_SHL1) || (Imm == DppCtrl::WAVE_ROL1) ||
10123 (Imm == DppCtrl::WAVE_SHR1) || (Imm == DppCtrl::WAVE_ROR1) ||
10124 (Imm == DppCtrl::ROW_MIRROR) || (Imm == DppCtrl::ROW_HALF_MIRROR) ||
10125 (Imm == DppCtrl::BCAST15) || (Imm == DppCtrl::BCAST31) ||
10126 (Imm >= DppCtrl::ROW_SHARE_FIRST &&
10127 Imm <= DppCtrl::ROW_SHARE_LAST) ||
10128 (Imm >= DppCtrl::ROW_XMASK_FIRST && Imm <= DppCtrl::ROW_XMASK_LAST);
10129 }
10130 return false;
10131}
10132
10133//===----------------------------------------------------------------------===//
10134// mAI
10135//===----------------------------------------------------------------------===//
10136
10137bool AMDGPUOperand::isBLGP() const {
10138 return isImm() && getImmTy() == ImmTyBLGP && isUInt<3>(getImm());
10139}
10140
10141bool AMDGPUOperand::isS16Imm() const {
10142 return isImmLiteral() && (isInt<16>(getImm()) || isUInt<16>(getImm()));
10143}
10144
10145bool AMDGPUOperand::isU16Imm() const {
10146 return isImmLiteral() && isUInt<16>(getImm());
10147}
10148
10149//===----------------------------------------------------------------------===//
10150// dim
10151//===----------------------------------------------------------------------===//
10152
10153bool AMDGPUAsmParser::parseDimId(unsigned &Encoding) {
10154 // We want to allow "dim:1D" etc.,
10155 // but the initial 1 is tokenized as an integer.
10156 std::string Token;
10157 if (isToken(AsmToken::Integer)) {
10158 SMLoc Loc = getToken().getEndLoc();
10159 Token = std::string(getTokenStr());
10160 lex();
10161 if (getLoc() != Loc)
10162 return false;
10163 }
10164
10165 StringRef Suffix;
10166 if (!parseId(Suffix))
10167 return false;
10168 Token += Suffix;
10169
10170 StringRef DimId = Token;
10171 DimId.consume_front("SQ_RSRC_IMG_");
10172
10173 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfoByAsmSuffix(DimId);
10174 if (!DimInfo)
10175 return false;
10176
10177 Encoding = DimInfo->Encoding;
10178 return true;
10179}
10180
10181ParseStatus AMDGPUAsmParser::parseDim(OperandVector &Operands) {
10182 if (!isGFX10Plus())
10183 return ParseStatus::NoMatch;
10184
10185 SMLoc S = getLoc();
10186
10187 if (!trySkipId("dim", AsmToken::Colon))
10188 return ParseStatus::NoMatch;
10189
10190 unsigned Encoding;
10191 SMLoc Loc = getLoc();
10192 if (!parseDimId(Encoding))
10193 return Error(Loc, "invalid dim value");
10194
10195 Operands.push_back(
10196 AMDGPUOperand::CreateImm(this, Encoding, S, AMDGPUOperand::ImmTyDim));
10197 return ParseStatus::Success;
10198}
10199
10200//===----------------------------------------------------------------------===//
10201// dpp
10202//===----------------------------------------------------------------------===//
10203
10204ParseStatus AMDGPUAsmParser::parseDPP8(OperandVector &Operands) {
10205 SMLoc S = getLoc();
10206
10207 if (!isGFX10Plus() || !trySkipId("dpp8", AsmToken::Colon))
10208 return ParseStatus::NoMatch;
10209
10210 // dpp8:[%d,%d,%d,%d,%d,%d,%d,%d]
10211
10212 int64_t Sels[8];
10213
10214 if (!skipToken(AsmToken::LBrac, "expected an opening square bracket"))
10215 return ParseStatus::Failure;
10216
10217 for (size_t i = 0; i < 8; ++i) {
10218 if (i > 0 && !skipToken(AsmToken::Comma, "expected a comma"))
10219 return ParseStatus::Failure;
10220
10221 SMLoc Loc = getLoc();
10222 if (getParser().parseAbsoluteExpression(Sels[i]))
10223 return ParseStatus::Failure;
10224 if (0 > Sels[i] || 7 < Sels[i])
10225 return Error(Loc, "expected a 3-bit value");
10226 }
10227
10228 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
10229 return ParseStatus::Failure;
10230
10231 unsigned DPP8 = 0;
10232 for (size_t i = 0; i < 8; ++i)
10233 DPP8 |= (Sels[i] << (i * 3));
10234
10235 Operands.push_back(
10236 AMDGPUOperand::CreateImm(this, DPP8, S, AMDGPUOperand::ImmTyDPP8));
10237 return ParseStatus::Success;
10238}
10239
10240bool AMDGPUAsmParser::isSupportedDPPCtrl(StringRef Ctrl,
10241 const OperandVector &Operands) {
10242 if (Ctrl == "row_newbcast")
10243 return isGFX90A();
10244
10245 if (Ctrl == "row_share" || Ctrl == "row_xmask")
10246 return isGFX10Plus();
10247
10248 if (Ctrl == "wave_shl" || Ctrl == "wave_shr" || Ctrl == "wave_rol" ||
10249 Ctrl == "wave_ror" || Ctrl == "row_bcast")
10250 return isVI() || isGFX9();
10251
10252 return Ctrl == "row_mirror" || Ctrl == "row_half_mirror" ||
10253 Ctrl == "quad_perm" || Ctrl == "row_shl" || Ctrl == "row_shr" ||
10254 Ctrl == "row_ror";
10255}
10256
10257int64_t AMDGPUAsmParser::parseDPPCtrlPerm() {
10258 // quad_perm:[%d,%d,%d,%d]
10259
10260 if (!skipToken(AsmToken::LBrac, "expected an opening square bracket"))
10261 return -1;
10262
10263 int64_t Val = 0;
10264 for (int i = 0; i < 4; ++i) {
10265 if (i > 0 && !skipToken(AsmToken::Comma, "expected a comma"))
10266 return -1;
10267
10268 int64_t Temp;
10269 SMLoc Loc = getLoc();
10270 if (getParser().parseAbsoluteExpression(Temp))
10271 return -1;
10272 if (Temp < 0 || Temp > 3) {
10273 Error(Loc, "expected a 2-bit value");
10274 return -1;
10275 }
10276
10277 Val += (Temp << i * 2);
10278 }
10279
10280 if (!skipToken(AsmToken::RBrac, "expected a closing square bracket"))
10281 return -1;
10282
10283 return Val;
10284}
10285
10286int64_t AMDGPUAsmParser::parseDPPCtrlSel(StringRef Ctrl) {
10287 using namespace AMDGPU::DPP;
10288
10289 // sel:%d
10290
10291 int64_t Val;
10292 SMLoc Loc = getLoc();
10293
10294 if (getParser().parseAbsoluteExpression(Val))
10295 return -1;
10296
10297 struct DppCtrlCheck {
10298 int64_t Ctrl;
10299 int Lo;
10300 int Hi;
10301 };
10302
10303 DppCtrlCheck Check =
10304 StringSwitch<DppCtrlCheck>(Ctrl)
10305 .Case("wave_shl", {DppCtrl::WAVE_SHL1, 1, 1})
10306 .Case("wave_rol", {DppCtrl::WAVE_ROL1, 1, 1})
10307 .Case("wave_shr", {DppCtrl::WAVE_SHR1, 1, 1})
10308 .Case("wave_ror", {DppCtrl::WAVE_ROR1, 1, 1})
10309 .Case("row_shl", {DppCtrl::ROW_SHL0, 1, 15})
10310 .Case("row_shr", {DppCtrl::ROW_SHR0, 1, 15})
10311 .Case("row_ror", {DppCtrl::ROW_ROR0, 1, 15})
10312 .Case("row_share", {DppCtrl::ROW_SHARE_FIRST, 0, 15})
10313 .Case("row_xmask", {DppCtrl::ROW_XMASK_FIRST, 0, 15})
10314 .Case("row_newbcast", {DppCtrl::ROW_NEWBCAST_FIRST, 0, 15})
10315 .Default({-1, 0, 0});
10316
10317 bool Valid;
10318 if (Check.Ctrl == -1) {
10319 Valid = (Ctrl == "row_bcast" && (Val == 15 || Val == 31));
10320 Val = (Val == 15) ? DppCtrl::BCAST15 : DppCtrl::BCAST31;
10321 } else {
10322 Valid = Check.Lo <= Val && Val <= Check.Hi;
10323 Val = (Check.Lo == Check.Hi) ? Check.Ctrl : (Check.Ctrl | Val);
10324 }
10325
10326 if (!Valid) {
10327 Error(Loc, Twine("invalid ", Ctrl) + Twine(" value"));
10328 return -1;
10329 }
10330
10331 return Val;
10332}
10333
10334ParseStatus AMDGPUAsmParser::parseDPPCtrl(OperandVector &Operands) {
10335 using namespace AMDGPU::DPP;
10336
10337 if (!isToken(AsmToken::Identifier) ||
10338 !isSupportedDPPCtrl(getTokenStr(), Operands))
10339 return ParseStatus::NoMatch;
10340
10341 SMLoc S = getLoc();
10342 int64_t Val = -1;
10343 StringRef Ctrl;
10344
10345 parseId(Ctrl);
10346
10347 if (Ctrl == "row_mirror") {
10348 Val = DppCtrl::ROW_MIRROR;
10349 } else if (Ctrl == "row_half_mirror") {
10350 Val = DppCtrl::ROW_HALF_MIRROR;
10351 } else {
10352 if (skipToken(AsmToken::Colon, "expected a colon")) {
10353 if (Ctrl == "quad_perm") {
10354 Val = parseDPPCtrlPerm();
10355 } else {
10356 Val = parseDPPCtrlSel(Ctrl);
10357 }
10358 }
10359 }
10360
10361 if (Val == -1)
10362 return ParseStatus::Failure;
10363
10364 Operands.push_back(
10365 AMDGPUOperand::CreateImm(this, Val, S, AMDGPUOperand::ImmTyDppCtrl));
10366 return ParseStatus::Success;
10367}
10368
10369void AMDGPUAsmParser::cvtVOP3DPP(MCInst &Inst, const OperandVector &Operands,
10370 bool IsDPP8) {
10371 OptionalImmIndexMap OptionalIdx;
10372 unsigned Opc = Inst.getOpcode();
10373 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10374
10375 // MAC instructions are special because they have 'old'
10376 // operand which is not tied to dst (but assumed to be).
10377 // They also have dummy unused src2_modifiers.
10378 int OldIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::old);
10379 int Src2ModIdx =
10380 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::src2_modifiers);
10381 bool IsMAC = OldIdx != -1 && Src2ModIdx != -1 &&
10382 Desc.getOperandConstraint(OldIdx, MCOI::TIED_TO) == -1;
10383
10384 unsigned I = 1;
10385 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10386 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10387 }
10388
10389 int Fi = 0;
10390 int VdstInIdx = AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::vdst_in);
10391 bool IsVOP3CvtSrDpp = Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx12 ||
10392 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx13 ||
10393 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx12 ||
10394 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx13 ||
10395 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx12 ||
10396 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx13 ||
10397 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx12 ||
10398 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx13;
10399
10400 for (unsigned E = Operands.size(); I != E; ++I) {
10401
10402 if (IsMAC) {
10403 int NumOperands = Inst.getNumOperands();
10404 if (OldIdx == NumOperands) {
10405 // Handle old operand
10406 constexpr int DST_IDX = 0;
10407 Inst.addOperand(Inst.getOperand(DST_IDX));
10408 } else if (Src2ModIdx == NumOperands) {
10409 // Add unused dummy src2_modifiers
10411 }
10412 }
10413
10414 if (VdstInIdx == static_cast<int>(Inst.getNumOperands())) {
10415 Inst.addOperand(Inst.getOperand(0));
10416 }
10417
10418 if (IsVOP3CvtSrDpp) {
10419 if (Src2ModIdx == static_cast<int>(Inst.getNumOperands())) {
10421 Inst.addOperand(MCOperand::createReg(MCRegister()));
10422 }
10423 }
10424
10425 auto TiedTo =
10426 Desc.getOperandConstraint(Inst.getNumOperands(), MCOI::TIED_TO);
10427 if (TiedTo != -1) {
10428 assert((unsigned)TiedTo < Inst.getNumOperands());
10429 // handle tied old or src2 for MAC instructions
10430 Inst.addOperand(Inst.getOperand(TiedTo));
10431 }
10432 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10433 // Add the register arguments
10434 if (IsDPP8 && Op.isDppFI()) {
10435 Fi = Op.getImm();
10436 } else if (isRegOrImmWithInputMods(Desc, Inst.getNumOperands())) {
10437 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10438 } else if (Op.isReg()) {
10439 Op.addRegOperands(Inst, 1);
10440 } else if (Op.isImm() &&
10441 Desc.operands()[Inst.getNumOperands()].RegClass != -1) {
10442 Op.addImmOperands(Inst, 1);
10443 } else if (Op.isImm()) {
10444 OptionalIdx[Op.getImmTy()] = I;
10445 } else {
10446 llvm_unreachable("unhandled operand type");
10447 }
10448 }
10449
10450 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp) && !IsVOP3CvtSrDpp)
10451 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10452 AMDGPUOperand::ImmTyClamp);
10453
10454 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::byte_sel)) {
10455 if (VdstInIdx == static_cast<int>(Inst.getNumOperands()))
10456 Inst.addOperand(Inst.getOperand(0));
10457 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10458 AMDGPUOperand::ImmTyByteSel);
10459 }
10460
10461 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
10462 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10463 AMDGPUOperand::ImmTyOModSI);
10464
10466 cvtVOP3P(Inst, Operands, OptionalIdx);
10467 else if (SIInstrFlags::isVOP3(Desc))
10468 cvtVOP3OpSel(Inst, Operands, OptionalIdx);
10469 else if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::op_sel)) {
10470 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10471 AMDGPUOperand::ImmTyOpSel);
10472 }
10473
10474 if (IsDPP8) {
10475 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10476 AMDGPUOperand::ImmTyDPP8);
10477 using namespace llvm::AMDGPU::DPP;
10478 Inst.addOperand(MCOperand::createImm(Fi ? DPP8_FI_1 : DPP8_FI_0));
10479 } else {
10480 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10481 AMDGPUOperand::ImmTyDppCtrl, 0xe4);
10482 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10483 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10484 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10485 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10486 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10487 AMDGPUOperand::ImmTyDppBoundCtrl);
10488
10489 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::fi))
10490 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10491 AMDGPUOperand::ImmTyDppFI);
10492 }
10493}
10494
10495void AMDGPUAsmParser::cvtDPP(MCInst &Inst, const OperandVector &Operands,
10496 bool IsDPP8) {
10497 OptionalImmIndexMap OptionalIdx;
10498
10499 unsigned I = 1;
10500 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10501 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10502 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10503 }
10504
10505 int Fi = 0;
10506 for (unsigned E = Operands.size(); I != E; ++I) {
10507 auto TiedTo =
10508 Desc.getOperandConstraint(Inst.getNumOperands(), MCOI::TIED_TO);
10509 if (TiedTo != -1) {
10510 assert((unsigned)TiedTo < Inst.getNumOperands());
10511 // handle tied old or src2 for MAC instructions
10512 Inst.addOperand(Inst.getOperand(TiedTo));
10513 }
10514 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10515 // Add the register arguments
10516 if (Op.isReg() && validateVccOperand(Op.getReg())) {
10517 // VOP2b (v_add_u32, v_sub_u32 ...) dpp use "vcc" token.
10518 // Skip it.
10519 continue;
10520 }
10521
10522 if (IsDPP8) {
10523 if (Op.isDPP8()) {
10524 Op.addImmOperands(Inst, 1);
10525 } else if (isRegOrImmWithInputMods(Desc, Inst.getNumOperands())) {
10526 Op.addRegWithFPInputModsOperands(Inst, 2);
10527 } else if (Op.isDppFI()) {
10528 Fi = Op.getImm();
10529 } else if (Op.isReg()) {
10530 Op.addRegOperands(Inst, 1);
10531 } else {
10532 llvm_unreachable("Invalid operand type");
10533 }
10534 } else {
10536 Op.addRegWithFPInputModsOperands(Inst, 2);
10537 } else if (Op.isReg()) {
10538 Op.addRegOperands(Inst, 1);
10539 } else if (Op.isDPPCtrl()) {
10540 Op.addImmOperands(Inst, 1);
10541 } else if (Op.isImm()) {
10542 // Handle optional arguments
10543 OptionalIdx[Op.getImmTy()] = I;
10544 } else {
10545 llvm_unreachable("Invalid operand type");
10546 }
10547 }
10548 }
10549
10550 if (IsDPP8) {
10551 using namespace llvm::AMDGPU::DPP;
10552 Inst.addOperand(MCOperand::createImm(Fi ? DPP8_FI_1 : DPP8_FI_0));
10553 } else {
10554 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10555 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10556 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10557 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10558 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10559 AMDGPUOperand::ImmTyDppBoundCtrl);
10560 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::fi)) {
10561 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10562 AMDGPUOperand::ImmTyDppFI);
10563 }
10564 }
10565}
10566
10567//===----------------------------------------------------------------------===//
10568// sdwa
10569//===----------------------------------------------------------------------===//
10570
10571ParseStatus AMDGPUAsmParser::parseSDWASel(OperandVector &Operands,
10572 StringRef Prefix,
10573 AMDGPUOperand::ImmTy Type) {
10574 return parseStringOrIntWithPrefix(
10575 Operands, Prefix,
10576 {"BYTE_0", "BYTE_1", "BYTE_2", "BYTE_3", "WORD_0", "WORD_1", "DWORD"},
10577 Type);
10578}
10579
10580ParseStatus AMDGPUAsmParser::parseSDWADstUnused(OperandVector &Operands) {
10581 return parseStringOrIntWithPrefix(
10582 Operands, "dst_unused", {"UNUSED_PAD", "UNUSED_SEXT", "UNUSED_PRESERVE"},
10583 AMDGPUOperand::ImmTySDWADstUnused);
10584}
10585
10586void AMDGPUAsmParser::cvtSdwaVOP1(MCInst &Inst, const OperandVector &Operands) {
10587 cvtSDWA(Inst, Operands, SDWAInstType::VOP1);
10588}
10589
10590void AMDGPUAsmParser::cvtSdwaVOP2(MCInst &Inst, const OperandVector &Operands) {
10591 cvtSDWA(Inst, Operands, SDWAInstType::VOP2);
10592}
10593
10594void AMDGPUAsmParser::cvtSdwaVOP2b(MCInst &Inst,
10595 const OperandVector &Operands) {
10596 cvtSDWA(Inst, Operands, SDWAInstType::VOP2, true, true);
10597}
10598
10599void AMDGPUAsmParser::cvtSdwaVOP2e(MCInst &Inst,
10600 const OperandVector &Operands) {
10601 cvtSDWA(Inst, Operands, SDWAInstType::VOP2, false, true);
10602}
10603
10604void AMDGPUAsmParser::cvtSdwaVOPC(MCInst &Inst, const OperandVector &Operands) {
10605 cvtSDWA(Inst, Operands, SDWAInstType::VOPC, isVI());
10606}
10607
10608void AMDGPUAsmParser::cvtSDWA(MCInst &Inst, const OperandVector &Operands,
10609 SDWAInstType BasicInstType, bool SkipDstVcc,
10610 bool SkipSrcVcc) {
10611 using namespace llvm::AMDGPU::SDWA;
10612
10613 OptionalImmIndexMap OptionalIdx;
10614 bool SkipVcc = SkipDstVcc || SkipSrcVcc;
10615 bool SkippedVcc = false;
10616
10617 unsigned I = 1;
10618 const MCInstrDesc &Desc = MII.get(Inst.getOpcode());
10619 for (unsigned J = 0; J < Desc.getNumDefs(); ++J) {
10620 ((AMDGPUOperand &)*Operands[I++]).addRegOperands(Inst, 1);
10621 }
10622
10623 for (unsigned E = Operands.size(); I != E; ++I) {
10624 AMDGPUOperand &Op = ((AMDGPUOperand &)*Operands[I]);
10625 if (SkipVcc && !SkippedVcc && Op.isReg() &&
10626 (Op.getReg() == AMDGPU::VCC || Op.getReg() == AMDGPU::VCC_LO)) {
10627 // VOP2b (v_add_u32, v_sub_u32 ...) sdwa use "vcc" token as dst.
10628 // Skip it if it's 2nd (e.g. v_add_i32_sdwa v1, vcc, v2, v3)
10629 // or 4th (v_addc_u32_sdwa v1, vcc, v2, v3, vcc) operand.
10630 // Skip VCC only if we didn't skip it on previous iteration.
10631 // Note that src0 and src1 occupy 2 slots each because of modifiers.
10632 if (BasicInstType == SDWAInstType::VOP2 &&
10633 ((SkipDstVcc && Inst.getNumOperands() == 1) ||
10634 (SkipSrcVcc && Inst.getNumOperands() == 5))) {
10635 SkippedVcc = true;
10636 continue;
10637 }
10638 if (BasicInstType == SDWAInstType::VOPC && Inst.getNumOperands() == 0) {
10639 SkippedVcc = true;
10640 continue;
10641 }
10642 }
10644 Op.addRegOrImmWithInputModsOperands(Inst, 2);
10645 } else if (Op.isImm()) {
10646 // Handle optional arguments
10647 OptionalIdx[Op.getImmTy()] = I;
10648 } else {
10649 llvm_unreachable("Invalid operand type");
10650 }
10651 SkippedVcc = false;
10652 }
10653
10654 const unsigned Opc = Inst.getOpcode();
10655 if (Opc != AMDGPU::V_NOP_sdwa_gfx10 && Opc != AMDGPU::V_NOP_sdwa_gfx9 &&
10656 Opc != AMDGPU::V_NOP_sdwa_vi) {
10657 // v_nop_sdwa_sdwa_vi/gfx9 has no optional sdwa arguments
10658 switch (BasicInstType) {
10659 case SDWAInstType::VOP1:
10660 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::clamp))
10661 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10662 AMDGPUOperand::ImmTyClamp, 0);
10663
10664 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::omod))
10665 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10666 AMDGPUOperand::ImmTyOModSI, 0);
10667
10668 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::dst_sel))
10669 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10670 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10671
10672 if (AMDGPU::hasNamedOperand(Opc, AMDGPU::OpName::dst_unused))
10673 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10674 AMDGPUOperand::ImmTySDWADstUnused,
10675 DstUnused::UNUSED_PRESERVE);
10676
10677 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10678 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10679 break;
10680
10681 case SDWAInstType::VOP2:
10682 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10683 AMDGPUOperand::ImmTyClamp, 0);
10684
10685 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::omod))
10686 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10687 AMDGPUOperand::ImmTyOModSI, 0);
10688
10689 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10690 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10691 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10692 AMDGPUOperand::ImmTySDWADstUnused,
10693 DstUnused::UNUSED_PRESERVE);
10694 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10695 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10696 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10697 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10698 break;
10699
10700 case SDWAInstType::VOPC:
10701 if (AMDGPU::hasNamedOperand(Inst.getOpcode(), AMDGPU::OpName::clamp))
10702 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10703 AMDGPUOperand::ImmTyClamp, 0);
10704 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10705 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10706 addOptionalImmOperand(Inst, Operands, OptionalIdx,
10707 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10708 break;
10709 }
10710 }
10711
10712 // special case v_mac_{f16, f32}:
10713 // it has src2 register operand that is tied to dst operand
10714 if (Inst.getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
10715 Inst.getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
10716 auto *it = Inst.begin();
10717 std::advance(
10718 it, AMDGPU::getNamedOperandIdx(Inst.getOpcode(), AMDGPU::OpName::src2));
10719 Inst.insert(it, Inst.getOperand(0)); // src2 = dst
10720 }
10721}
10722
10723/// Force static initialization.
10724extern "C" LLVM_ABI LLVM_EXTERNAL_VISIBILITY void
10730
10731#define GET_MATCHER_IMPLEMENTATION
10732#define GET_MNEMONIC_SPELL_CHECKER
10733#define GET_MNEMONIC_CHECKER
10734#include "AMDGPUGenAsmMatcher.inc"
10735
10736ParseStatus AMDGPUAsmParser::parseCustomOperand(OperandVector &Operands,
10737 unsigned MCK) {
10738 switch (MCK) {
10739 case MCK_addr64:
10740 return parseTokenOp("addr64", Operands);
10741 case MCK_done:
10742 return parseNamedBit("done", Operands, AMDGPUOperand::ImmTyDone, true);
10743 case MCK_idxen:
10744 return parseTokenOp("idxen", Operands);
10745 case MCK_lds:
10746 return parseNamedBit("lds", Operands, AMDGPUOperand::ImmTyLDS,
10747 /*IgnoreNegative=*/true);
10748 case MCK_offen:
10749 return parseTokenOp("offen", Operands);
10750 case MCK_off:
10751 return parseTokenOp("off", Operands);
10752 case MCK_row_95_en:
10753 return parseNamedBit("row_en", Operands, AMDGPUOperand::ImmTyRowEn, true);
10754 case MCK_gds:
10755 return parseNamedBit("gds", Operands, AMDGPUOperand::ImmTyGDS);
10756 case MCK_tfe:
10757 return parseNamedBit("tfe", Operands, AMDGPUOperand::ImmTyTFE);
10758 }
10759 return tryCustomParseOperand(Operands, MCK);
10760}
10761
10762// This function should be defined after auto-generated include so that we have
10763// MatchClassKind enum defined
10764unsigned AMDGPUAsmParser::validateTargetOperandClass(MCParsedAsmOperand &Op,
10765 unsigned Kind) {
10766 // Tokens like "glc" would be parsed as immediate operands in ParseOperand().
10767 // But MatchInstructionImpl() expects to meet token and fails to validate
10768 // operand. This method checks if we are given immediate operand but expect to
10769 // get corresponding token.
10770 AMDGPUOperand &Operand = (AMDGPUOperand &)Op;
10771 switch (Kind) {
10772 case MCK_addr64:
10773 return Operand.isAddr64() ? Match_Success : Match_InvalidOperand;
10774 case MCK_gds:
10775 return Operand.isGDS() ? Match_Success : Match_InvalidOperand;
10776 case MCK_lds:
10777 return Operand.isLDS() ? Match_Success : Match_InvalidOperand;
10778 case MCK_idxen:
10779 return Operand.isIdxen() ? Match_Success : Match_InvalidOperand;
10780 case MCK_offen:
10781 return Operand.isOffen() ? Match_Success : Match_InvalidOperand;
10782 case MCK_tfe:
10783 return Operand.isTFE() ? Match_Success : Match_InvalidOperand;
10784 case MCK_done:
10785 return Operand.isDone() ? Match_Success : Match_InvalidOperand;
10786 case MCK_row_95_en:
10787 return Operand.isRowEn() ? Match_Success : Match_InvalidOperand;
10788 case MCK_SSrc_b32:
10789 // When operands have expression values, they will return true for isToken,
10790 // because it is not possible to distinguish between a token and an
10791 // expression at parse time. MatchInstructionImpl() will always try to
10792 // match an operand as a token, when isToken returns true, and when the
10793 // name of the expression is not a valid token, the match will fail,
10794 // so we need to handle it here.
10795 return Operand.isSSrc_b32() ? Match_Success : Match_InvalidOperand;
10796 case MCK_SSrc_f32:
10797 return Operand.isSSrc_f32() ? Match_Success : Match_InvalidOperand;
10798 case MCK_SOPPBrTarget:
10799 return Operand.isSOPPBrTarget() ? Match_Success : Match_InvalidOperand;
10800 case MCK_VReg32OrOff:
10801 return Operand.isVReg32OrOff() ? Match_Success : Match_InvalidOperand;
10802 case MCK_InterpSlot:
10803 return Operand.isInterpSlot() ? Match_Success : Match_InvalidOperand;
10804 case MCK_InterpAttr:
10805 return Operand.isInterpAttr() ? Match_Success : Match_InvalidOperand;
10806 case MCK_InterpAttrChan:
10807 return Operand.isInterpAttrChan() ? Match_Success : Match_InvalidOperand;
10808 case MCK_SReg_64:
10809 case MCK_SReg_64_XEXEC:
10810 // Null is defined as a 32-bit register but
10811 // it should also be enabled with 64-bit operands or larger.
10812 // The following code enables it for SReg_64 and larger operands
10813 // used as source and destination. Remaining source
10814 // operands are handled in isInlinableImm.
10815 case MCK_SReg_96:
10816 case MCK_SReg_128:
10817 case MCK_SReg_256:
10818 case MCK_SReg_512:
10819 return Operand.isNull() ? Match_Success : Match_InvalidOperand;
10820 default:
10821 return Match_InvalidOperand;
10822 }
10823}
10824
10825//===----------------------------------------------------------------------===//
10826// endpgm
10827//===----------------------------------------------------------------------===//
10828
10829ParseStatus AMDGPUAsmParser::parseEndpgm(OperandVector &Operands) {
10830 SMLoc S = getLoc();
10831 int64_t Imm = 0;
10832
10833 if (!parseExpr(Imm)) {
10834 // The operand is optional, if not present default to 0
10835 Imm = 0;
10836 }
10837
10838 if (!isUInt<16>(Imm))
10839 return Error(S, "expected a 16-bit value");
10840
10841 Operands.push_back(
10842 AMDGPUOperand::CreateImm(this, Imm, S, AMDGPUOperand::ImmTyEndpgm));
10843 return ParseStatus::Success;
10844}
10845
10846bool AMDGPUOperand::isEndpgm() const { return isImmTy(ImmTyEndpgm); }
10847
10848//===----------------------------------------------------------------------===//
10849// Split Barrier
10850//===----------------------------------------------------------------------===//
10851
10852bool AMDGPUOperand::isSplitBarrier() const {
10853 if (!isImm())
10854 return false;
10855
10856 int64_t Imm = getImm();
10859}
#define Success
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
SmallVector< int16_t, MAX_SRC_OPERANDS_NUM > OperandIndices
static bool checkWriteLane(const MCInst &Inst)
static bool getRegNum(StringRef Str, unsigned &Num)
static void addSrcModifiersAndSrc(MCInst &Inst, const OperandVector &Operands, unsigned i, unsigned Opc, AMDGPU::OpName OpName)
static constexpr RegInfo RegularRegisters[]
static const RegInfo * getRegularRegInfo(StringRef Str)
static ArrayRef< unsigned > getAllVariants()
static OperandIndices getSrcOperandIndices(unsigned Opcode, bool AddMandatoryLiterals=false)
static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name, const MCRegisterInfo *MRI)
static bool IsMovrelsSDWAOpcode(const unsigned Opcode)
static const fltSemantics * getFltSemantics(unsigned Size)
static bool isRegularReg(RegisterKind Kind)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUAsmParser()
Force static initialization.
static bool ConvertOmodMul(int64_t &Mul)
#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE)
static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi)
static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT)
static bool AMDGPUCheckMnemonic(StringRef Mnemonic, const FeatureBitset &AvailableFeatures, unsigned VariantID)
static void applyMnemonicAliases(StringRef &Mnemonic, const FeatureBitset &Features, unsigned VariantID)
constexpr unsigned MAX_SRC_OPERANDS_NUM
#define EXPR_RESOLVE_OR_ERROR(RESOLVED)
static bool ConvertOmodDiv(int64_t &Div)
static bool IsRevOpcode(const unsigned Opcode)
static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal, int64_t CntVal, bool Saturate, unsigned(*encode)(const IsaVersion &Version, unsigned, unsigned), unsigned(*decode)(const IsaVersion &Version, unsigned))
static MCRegister getSpecialRegForName(StringRef RegName)
static void addOptionalImmOperand(MCInst &Inst, const OperandVector &Operands, AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx, AMDGPUOperand::ImmTy ImmT, int64_t Default=0, std::optional< unsigned > InsertAt=std::nullopt)
static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI)
static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum)
static const fltSemantics * getOpFltSemantics(uint8_t OperandType)
static bool isInvalidVOPDY(const OperandVector &Operands, uint64_t InvalidOprIdx)
static std::string AMDGPUMnemonicSpellCheck(StringRef S, const FeatureBitset &FBS, unsigned VariantID=0)
static LLVM_READNONE unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask, const unsigned XorMask)
static bool isSafeTruncation(int64_t Val, unsigned Size)
unsigned uint64_t
AMDHSA kernel descriptor MCExpr struct for use in MC layer.
Provides AMDGPU specific target descriptions.
AMDGPU metadata definitions and in-memory representations.
Enums shared between the AMDGPU backend (LLVM) and the ELF linker (LLD) for the .amdgpu....
AMDHSA kernel descriptor definitions.
static bool parseExpr(MCAsmParser &MCParser, const MCExpr *&Value, raw_ostream &Err)
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file declares a class to represent arbitrary precision floating point values and provide a varie...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READNONE
Definition Compiler.h:331
#define LLVM_ABI
Definition Compiler.h:215
#define LLVM_EXTERNAL_VISIBILITY
Definition Compiler.h:132
@ Default
#define Check(C,...)
static llvm::Expected< InlineInfo > decode(GsymDataExtractor &Data, uint64_t &Offset, uint64_t BaseAddr)
Decode an InlineInfo in Data at the specified offset.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
Loop::LoopBounds::Direction Direction
Definition LoopInfo.cpp:253
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
#define P(N)
if(PassOpts->AAPipeline)
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Fold Operands
Interface definition for SIInstrInfo.
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
const char * Msg
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static void initialize(TargetLibraryInfoImpl &TLI, const Triple &T, const llvm::StringTable &StandardNames, VectorLibrary VecLib)
Initialize the set of available library functions based on the specified target triple.
BinaryOperator * Mul
static const char * getRegisterName(MCRegister Reg)
static const AMDGPUMCExpr * createMax(ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static unsigned getNumExpectedArgs(VariantKind Kind)
static const AMDGPUMCExpr * createLit(LitModifier Lit, int64_t Value, MCContext &Ctx)
static const AMDGPUMCExpr * create(VariantKind Kind, ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createExtraSGPRs(const MCExpr *VCCUsed, const MCExpr *FlatScrUsed, bool XNACKUsed, MCContext &Ctx)
Allow delayed MCExpr resolve of ExtraSGPRs (in case VCCUsed or FlatScrUsed are unresolvable but neede...
static const AMDGPUMCExpr * createAlignTo(const MCExpr *Value, const MCExpr *Align, MCContext &Ctx)
static std::optional< TargetID > parseTargetIDString(StringRef TargetIDDirective)
Parse and validate a TargetID from a full "<triple>-<processor>:<features>" directive string.
TargetIDSetting getXnackSetting() const
StringRef getTargetTripleString() const
std::string toString() const
TargetIDSetting getSramEccSetting() const
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & BFloat()
Definition APFloat.h:303
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
opStatus
IEEE-754R 7: Default exception handling.
Definition APFloat.h:377
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:6034
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
Definition ArrayRef.h:218
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
Definition ArrayRef.h:194
const T & front() const
Get the first element.
Definition ArrayRef.h:144
iterator end() const
Definition ArrayRef.h:130
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
StringRef getString() const
Get the string for the current token, this includes all characters (for example, the quotes on string...
Definition MCAsmMacro.h:103
bool is(TokenKind K) const
Definition MCAsmMacro.h:75
Register getReg() const
Container class for subtarget features.
constexpr bool test(unsigned I) const
constexpr FeatureBitset & flip(unsigned I)
void printExpr(raw_ostream &, const MCExpr &) const
virtual void Initialize(MCAsmParser &Parser)
Initialize the extension for parsing using the given Parser.
static const MCBinaryExpr * createAdd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:342
static const MCBinaryExpr * createDiv(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:352
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
Definition MCExpr.h:427
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Definition MCExpr.cpp:212
Context object for machine code objects.
Definition MCContext.h:83
LLVM_ABI MCSymbol * getOrCreateSymbol(const Twine &Name)
Lookup the symbol inside with the specified Name.
Instances of this class represent a single low-level machine instruction.
Definition MCInst.h:188
unsigned getNumOperands() const
Definition MCInst.h:212
SMLoc getLoc() const
Definition MCInst.h:208
void setLoc(SMLoc loc)
Definition MCInst.h:207
unsigned getOpcode() const
Definition MCInst.h:202
iterator insert(iterator I, const MCOperand &Op)
Definition MCInst.h:232
void addOperand(const MCOperand Op)
Definition MCInst.h:215
iterator begin()
Definition MCInst.h:227
size_t size() const
Definition MCInst.h:226
const MCOperand & getOperand(unsigned i) const
Definition MCInst.h:210
Describe properties that are true of each instruction in the target description file.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
const int16_t * getRegClassByHwModeTable(unsigned ModeId) const
Definition MCInstrInfo.h:70
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Definition MCInstrInfo.h:79
Instances of this class represent operands of the MCInst class.
Definition MCInst.h:40
void setImm(int64_t Val)
Definition MCInst.h:89
static MCOperand createExpr(const MCExpr *Val)
Definition MCInst.h:166
int64_t getImm() const
Definition MCInst.h:84
static MCOperand createReg(MCRegister Reg)
Definition MCInst.h:138
static MCOperand createImm(int64_t Val)
Definition MCInst.h:145
bool isImm() const
Definition MCInst.h:66
void setReg(MCRegister Reg)
Set the register number.
Definition MCInst.h:79
bool isReg() const
Definition MCInst.h:65
MCRegister getReg() const
Returns the register number.
Definition MCInst.h:73
const MCExpr * getExpr() const
Definition MCInst.h:118
bool isExpr() const
Definition MCInst.h:69
MCParsedAsmOperand - This abstract class represents a source-level assembly instruction operand.
MCRegisterClass - Base class of TargetRegisterClass.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
constexpr bool isValid() const
Definition MCRegister.h:84
Generic base class for all target subtargets.
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
Definition MCSymbol.h:42
StringRef getName() const
getName - Get the symbol name.
Definition MCSymbol.h:188
bool isVariable() const
isVariable - Check if this is a variable symbol.
Definition MCSymbol.h:267
LLVM_ABI void setVariableValue(const MCExpr *Value)
Definition MCSymbol.cpp:50
void setRedefinable(bool Value)
Mark this symbol as redefinable.
Definition MCSymbol.h:210
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
Definition MCSymbol.h:270
MCTargetAsmParser - Generic interface to target specific assembly parsers.
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Ternary parse status returned by various parse* methods.
constexpr bool isFailure() const
static constexpr StatusTy Failure
constexpr bool isSuccess() const
static constexpr StatusTy Success
static constexpr StatusTy NoMatch
constexpr bool isNoMatch() const
constexpr unsigned id() const
Definition Register.h:100
Represents a location in source code.
Definition SMLoc.h:22
static SMLoc getFromPointer(const char *Ptr)
Definition SMLoc.h:35
constexpr const char * getPointer() const
Definition SMLoc.h:33
constexpr bool isValid() const
Definition SMLoc.h:28
SMLoc Start
Definition SMLoc.h:49
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Definition StringRef.h:888
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool consume_back(StringRef Suffix)
Returns true if this StringRef has the given suffix and removes that suffix.
Definition StringRef.h:691
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
Definition StringRef.h:597
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
Definition StringRef.h:635
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
Definition StringRef.h:138
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
Definition StringRef.h:270
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
Definition StringRef.h:661
bool contains(StringRef key) const
Check if the set contains the given key.
Definition StringSet.h:60
std::pair< typename Base::iterator, bool > insert(StringRef key)
Definition StringSet.h:39
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
Definition Twine.cpp:17
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
This class implements an extremely fast bulk output stream that can only output to a stream.
Definition raw_ostream.h:53
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
unsigned getTgtId(const StringRef Name)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumSGPRs[]
Key for Kernel::CodeProps::Metadata::mNumSGPRs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
constexpr char AssemblerDirectiveBegin[]
HSA metadata beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
HSA metadata ending assembler directive.
constexpr char AssemblerDirectiveBegin[]
Old HSA metadata beginning assembler directive for V2.
int64_t getHwregId(StringRef Name, const MCSubtargetInfo &STI)
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs)
Returns true if Lhs and Rhs are incompatible (both specific but different).
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
unsigned getDefaultFormatEncoding(const MCSubtargetInfo &STI)
int64_t convertDfmtNfmt2Ufmt(unsigned Dfmt, unsigned Nfmt, const MCSubtargetInfo &STI)
int64_t encodeDfmtNfmt(unsigned Dfmt, unsigned Nfmt)
int64_t getUnifiedFormat(const StringRef Name, const MCSubtargetInfo &STI)
bool isValidFormatEncoding(unsigned Val, const MCSubtargetInfo &STI)
int64_t getNfmt(const StringRef Name, const MCSubtargetInfo &STI)
int64_t getDfmt(const StringRef Name)
constexpr char AssemblerDirective[]
PAL metadata (old linear format) assembler directive.
constexpr char AssemblerDirectiveBegin[]
PAL metadata (new MsgPack format) beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
PAL metadata (new MsgPack format) ending assembler directive.
int64_t getMsgOpId(int64_t MsgId, StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a sendmsg operation to the operation portion of the immediate encoding.
int64_t getMsgId(StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a msg_id to the message portion of the immediate encoding.
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
ArrayRef< GFXVersion > getGFXVersions()
constexpr unsigned COMPONENTS[]
constexpr const char *const ModMatrixFmt[]
constexpr const char *const ModMatrixScaleFmt[]
constexpr const char *const ModMatrixScale[]
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isInlineValue(MCRegister Reg)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
FuncInfoFlags
Per-function flags packed into INFO_FLAGS entries.
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const int OPR_ID_UNSUPPORTED
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getTemporalHintType(const MCInstrDesc TID)
LLVM_READONLY bool isLitExpr(const MCExpr *Expr)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_ABI bool isCPUValidForSubArch(Triple::SubArchType SubArch, GPUKind AK)
Return true if the GPU AK is usable with the triple subarch SubArch.
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
bool hasPackedD16(const MCSubtargetInfo &STI)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
bool isSI(const MCSubtargetInfo &STI)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
bool isGFX9(const MCSubtargetInfo &STI)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
GPUKind
GPU kinds supported by the AMDGPU target.
bool isGFX90A(const MCSubtargetInfo &STI)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool hasMAIInsts(const MCSubtargetInfo &STI)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByAsmSuffix(StringRef AsmSuffix)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
LLVM_READONLY int64_t getLitValue(const MCExpr *Expr)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
bool isGFX10Plus(const MCSubtargetInfo &STI)
AMDGPU::TargetID TargetID
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_REG_IMM_V2FP64
Definition SIDefines.h:441
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
Definition SIDefines.h:459
@ OPERAND_REG_IMM_INT64
Definition SIDefines.h:426
@ OPERAND_REG_IMM_V2FP16
Definition SIDefines.h:434
@ OPERAND_REG_INLINE_C_FP64
Definition SIDefines.h:450
@ OPERAND_REG_IMM_NOINLINE_FP16
Definition SIDefines.h:432
@ OPERAND_REG_INLINE_C_BF16
Definition SIDefines.h:447
@ OPERAND_REG_INLINE_C_V2BF16
Definition SIDefines.h:452
@ OPERAND_REG_IMM_V2INT64
Definition SIDefines.h:437
@ OPERAND_REG_IMM_V2INT16
Definition SIDefines.h:436
@ OPERAND_REG_IMM_BF16
Definition SIDefines.h:430
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
Definition SIDefines.h:425
@ OPERAND_REG_IMM_V2BF16
Definition SIDefines.h:433
@ OPERAND_REG_IMM_FP16
Definition SIDefines.h:431
@ OPERAND_REG_IMM_V2FP16_SPLAT
Definition SIDefines.h:435
@ OPERAND_REG_INLINE_C_INT64
Definition SIDefines.h:446
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
Definition SIDefines.h:444
@ OPERAND_REG_IMM_NOINLINE_V2FP16
Definition SIDefines.h:438
@ OPERAND_REG_IMM_FP64
Definition SIDefines.h:429
@ OPERAND_REG_INLINE_C_V2FP16
Definition SIDefines.h:453
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
Definition SIDefines.h:464
@ OPERAND_REG_INLINE_AC_FP32
Definition SIDefines.h:465
@ OPERAND_REG_IMM_V2INT32
Definition SIDefines.h:439
@ OPERAND_REG_IMM_FP32
Definition SIDefines.h:428
@ OPERAND_REG_INLINE_C_FP32
Definition SIDefines.h:449
@ OPERAND_REG_INLINE_C_INT32
Definition SIDefines.h:445
@ OPERAND_REG_INLINE_C_V2INT16
Definition SIDefines.h:451
@ OPERAND_REG_IMM_V2FP32
Definition SIDefines.h:440
@ OPERAND_REG_INLINE_AC_FP64
Definition SIDefines.h:466
@ OPERAND_REG_INLINE_C_FP16
Definition SIDefines.h:448
@ OPERAND_REG_IMM_INT16
Definition SIDefines.h:427
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
Definition SIDefines.h:456
constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo)
Is this a scalar (i.e. not packed) bf16 source operand?
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
LLVM_ABI StringRef getArchNameAMDGCN(GPUKind AK)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
bool isMAC(unsigned Opc)
LLVM_READNONE unsigned getOperandSize(const MCOperandInfo &OpInfo)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const int OPR_ID_UNKNOWN
bool isGFX1250Plus(const MCSubtargetInfo &STI)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
bool isPermlane16(unsigned Opc)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ STT_AMDGPU_HSA_KERNEL
Definition ELF.h:1447
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ OPERAND_IMMEDIATE
Definition MCInstrDesc.h:61
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
void validate(const Triple &TT, const FeatureBitset &FeatureBits)
constexpr bool isAtomicRet(const T &...O)
Definition SIDefines.h:361
constexpr bool isVOPC(const T &...O)
Definition SIDefines.h:233
constexpr bool isVOP3(const T &...O)
Definition SIDefines.h:236
constexpr bool isVOP1(const T &...O)
Definition SIDefines.h:227
constexpr bool usesTENSOR_CNT(const T &...O)
Definition SIDefines.h:307
constexpr bool isMAI(const T &...O)
Definition SIDefines.h:349
constexpr bool isVOP2(const T &...O)
Definition SIDefines.h:230
constexpr bool isSWMMAC(const T &...O)
Definition SIDefines.h:376
constexpr bool isSOP2(const T &...O)
Definition SIDefines.h:215
constexpr bool isFLAT(const T &...O)
Definition SIDefines.h:283
constexpr bool isVOP3P(const T &...O)
Definition SIDefines.h:239
constexpr bool isBuffer(const T &...O)
Definition SIDefines.h:264
constexpr bool hasIntClamp(const T &...O)
Definition SIDefines.h:328
constexpr bool isAtomicNoRet(const T &...O)
Definition SIDefines.h:358
constexpr bool isSMRD(const T &...O)
Definition SIDefines.h:268
constexpr bool isVOP3Like(const T &...O)
Definition SIDefines.h:242
constexpr bool isMIMG(const T &...O)
Definition SIDefines.h:271
constexpr bool isVMEM(const T &...O)
Definition SIDefines.h:401
constexpr bool isImage(const T &...O)
Definition SIDefines.h:397
constexpr bool isWMMA(const T &...O)
Definition SIDefines.h:364
constexpr bool isVOPD3(const T &...O)
Definition SIDefines.h:379
constexpr bool isGWS(const T &...O)
Definition SIDefines.h:373
constexpr bool isMUBUF(const T &...O)
Definition SIDefines.h:258
constexpr bool isSDWA(const T &...O)
Definition SIDefines.h:249
constexpr bool isSOPC(const T &...O)
Definition SIDefines.h:218
constexpr bool isDOT(const T &...O)
Definition SIDefines.h:352
constexpr bool isVSAMPLE(const T &...O)
Definition SIDefines.h:277
constexpr bool isDS(const T &...O)
Definition SIDefines.h:286
constexpr bool isAtomic(const T &...O)
Definition SIDefines.h:390
constexpr bool isGather4(const T &...O)
Definition SIDefines.h:304
constexpr bool isPacked(const T &...O)
Definition SIDefines.h:337
constexpr bool isDPP(const T &...O)
Definition SIDefines.h:252
constexpr bool isSegmentSpecificFLAT(const T &...O)
Definition SIDefines.h:393
@ Valid
The data is already valid.
EnumSet< Modifier > Modifiers
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
bool isNull(StringRef S)
Definition YAMLTraits.h:571
This is an optimization pass for GlobalISel generic memory operations.
bool errorToBool(Error Err)
Helper for converting an Error to a bool.
Definition Error.h:1129
@ Offset
Definition DWP.cpp:577
StringMapEntry< Value * > ValueName
Definition Value.h:56
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
unsigned encode(MaybeAlign A)
Returns a representation of the alignment that encodes undefined as 0.
Definition Alignment.h:206
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
static bool isMem(const MachineInstr &MI, unsigned Op)
LLVM_ABI std::pair< StringRef, StringRef > getToken(StringRef Source, StringRef Delimiters=" \t\n\v\f\r")
getToken - This function extracts one token from source, ignoring any leading characters that appear ...
static StringRef getCPU(StringRef CPU)
Processes a CPU name.
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
LLVM_ABI void PrintError(const Twine &Msg)
Definition Error.cpp:104
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Definition bit.h:362
Op::Description Desc
Target & getTheR600Target()
The target for R600 GPUs.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
SmallVectorImpl< std::unique_ptr< MCParsedAsmOperand > > OperandVector
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
Target & getTheGCNTarget()
The target for GCN GPUs.
@ Sub
Subtraction of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
unsigned M0(unsigned Val)
Definition VE.h:376
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
Definition DWP.h:31
#define N
RegisterKind Kind
StringLiteral Name
void initDefault(const MCSubtargetInfo &STI, MCContext &Ctx, bool InitMCExpr=true)
void validate(const MCSubtargetInfo *STI, MCContext &Ctx)
SmallVector< std::pair< MCSymbol *, std::string >, 4 > IndirectCalls
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 8 > Calls
SmallVector< FuncInfo, 8 > Funcs
SmallVector< std::pair< MCSymbol *, std::string >, 4 > TypeIds
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 4 > Uses
Instruction set architecture version.
static void bits_set(const MCExpr *&Dst, const MCExpr *Value, uint32_t Shift, uint32_t Mask, MCContext &Ctx)
static MCKernelDescriptor getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx)
RegisterMCAsmParser - Helper template for registering a target specific assembly parser,...