70 enum KindTy { Token, Immediate, Register, Expression } Kind;
72 SMLoc StartLoc, EndLoc;
73 const AMDGPUAsmParser *AsmParser;
76 AMDGPUOperand(KindTy Kind_,
const AMDGPUAsmParser *AsmParser_)
77 : Kind(Kind_), AsmParser(AsmParser_) {}
79 using Ptr = std::unique_ptr<AMDGPUOperand>;
87 bool hasFPModifiers()
const {
return Abs || Neg; }
88 bool hasIntModifiers()
const {
return Sext; }
89 bool hasModifiers()
const {
return hasFPModifiers() || hasIntModifiers(); }
90 bool isForcedLit()
const {
return Lit == LitModifier::Lit; }
91 bool isForcedLit64()
const {
return Lit == LitModifier::Lit64; }
93 int64_t getFPModifiersOperand()
const {
100 int64_t getIntModifiersOperand()
const {
106 int64_t getModifiersOperand()
const {
107 assert(!(hasFPModifiers() && hasIntModifiers()) &&
108 "fp and int modifiers should not be used simultaneously");
109 if (hasFPModifiers())
110 return getFPModifiersOperand();
111 if (hasIntModifiers())
112 return getIntModifiersOperand();
116 friend raw_ostream &
operator<<(raw_ostream &OS,
117 AMDGPUOperand::Modifiers Mods);
191 ImmTyMatrixAScaleFmt,
192 ImmTyMatrixBScaleFmt,
225 mutable int MCOpIdx = -1;
228 bool isToken()
const override {
return Kind == Token; }
230 bool isSymbolRefExpr()
const {
234 bool isImm()
const override {
return Kind == Immediate; }
236 bool isInlinableImm(MVT type)
const;
237 bool isLiteralImm(MVT type)
const;
239 bool isRegKind()
const {
return Kind == Register; }
241 bool isReg()
const override {
return isRegKind() && !hasModifiers(); }
243 bool isRegOrInline(
unsigned RCID, MVT type)
const {
244 return isRegClass(RCID) || isInlinableImm(type);
247 bool isRegOrInlineTarget(
unsigned TargetRCIdx, MVT type)
const {
248 return isRegClassTarget(TargetRCIdx) || isInlinableImm(type);
252 return isRegOrInline(RCID, type) || isLiteralImm(type);
255 bool isRegOrImmWithInputModsTarget(
unsigned TargetRCIdx, MVT type)
const {
256 return isRegOrInlineTarget(TargetRCIdx, type) || isLiteralImm(type);
259 template <
bool IsFake16>
bool isRegOrImmWithIntT16InputMods()
const {
261 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
264 bool isRegOrImmWithInt32InputMods()
const {
268 template <
bool IsFake16>
bool isRegOrInlineImmWithIntT16InputMods()
const {
269 return isRegOrInline(
270 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::i16);
273 bool isRegOrInlineImmWithInt32InputMods()
const {
274 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::i32);
277 bool isRegOrImmWithInt64InputMods()
const {
278 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
281 bool isRegOrImmWithFP16InputMods()
const {
285 template <
bool IsFake16>
bool isRegOrImmWithFPT16InputMods()
const {
287 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
290 bool isRegOrImmWithFPT16_LO16InputMods()
const {
294 bool isRegOrImmWithFP32InputMods()
const {
298 bool isRegOrImmWithFP64InputMods()
const {
299 return isRegOrImmWithInputModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
302 template <
bool IsFake16>
bool isRegOrInlineImmWithFP16InputMods()
const {
303 return isRegOrInline(
304 IsFake16 ? AMDGPU::VS_32RegClassID : AMDGPU::VS_16RegClassID, MVT::f16);
307 bool isRegOrInlineImmWithFP32InputMods()
const {
308 return isRegOrInline(AMDGPU::VS_32RegClassID, MVT::f32);
311 bool isRegOrInlineImmWithFP64InputMods()
const {
312 return isRegOrInlineTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
315 bool isVRegWithInputMods(
unsigned RCID)
const {
return isRegClass(RCID); }
317 bool isVRegWithFP32InputMods()
const {
318 return isVRegWithInputMods(AMDGPU::VGPR_32RegClassID);
321 bool isVRegWithFP64InputMods()
const {
322 return isRegClassTarget(AMDGPU::VReg_64_AlignTarget);
325 bool isPackedFP16InputMods()
const {
329 bool isPackedVGPRFP32InputMods()
const {
333 bool isVReg32()
const {
return isRegClass(AMDGPU::VGPR_32RegClassID); }
335 bool isVReg32OrOff()
const {
return isOff() || isVReg32(); }
337 bool isRsrcReg32()
const {
return isRegClass(AMDGPU::RsrcReg32RegClassID); }
339 bool isNull()
const {
return isRegKind() &&
getReg() == AMDGPU::SGPR_NULL; }
341 bool isAV_LdSt_32_Align2_RegOp()
const {
342 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
343 isRegClass(AMDGPU::AGPR_32RegClassID);
346 bool isVRegWithInputMods()
const;
347 template <
bool IsFake16>
bool isT16_Lo128VRegWithInputMods()
const;
348 template <
bool IsFake16>
bool isT16VRegWithInputMods()
const;
349 bool isT16_LO16VRegWithInputMods()
const;
351 bool isSDWAOperand(MVT type)
const;
352 bool isSDWAFP16Operand()
const;
353 bool isSDWAFP32Operand()
const;
354 bool isSDWAInt16Operand()
const;
355 bool isSDWAInt32Operand()
const;
357 bool isImmTy(ImmTy ImmT)
const {
return isImm() &&
Imm.Type == ImmT; }
359 template <ImmTy Ty>
bool isImmTy()
const {
return isImmTy(Ty); }
361 bool isImmLiteral()
const {
return isImmTy(ImmTyNone); }
363 bool isImmModifier()
const {
return isImm() &&
Imm.Type != ImmTyNone; }
365 bool isOModSI()
const {
return isImmTy(ImmTyOModSI); }
366 bool isDim()
const {
return isImmTy(ImmTyDim); }
367 bool isR128A16()
const {
return isImmTy(ImmTyR128A16); }
368 bool isOff()
const {
return isImmTy(ImmTyOff); }
369 bool isExpTgt()
const {
return isImmTy(ImmTyExpTgt); }
370 bool isOffen()
const {
return isImmTy(ImmTyOffen); }
371 bool isIdxen()
const {
return isImmTy(ImmTyIdxen); }
372 bool isAddr64()
const {
return isImmTy(ImmTyAddr64); }
373 bool isSMEMOffsetMod()
const {
return isImmTy(ImmTySMEMOffsetMod); }
374 bool isFlatOffset()
const {
375 return isImmTy(ImmTyOffset) || isImmTy(ImmTyInstOffset);
377 bool isGDS()
const {
return isImmTy(ImmTyGDS); }
378 bool isLDS()
const {
return isImmTy(ImmTyLDS); }
379 bool isCPol()
const {
return isImmTy(ImmTyCPol); }
380 bool isIndexKey8bit()
const {
return isImmTy(ImmTyIndexKey8bit); }
381 bool isIndexKey16bit()
const {
return isImmTy(ImmTyIndexKey16bit); }
382 bool isIndexKey32bit()
const {
return isImmTy(ImmTyIndexKey32bit); }
383 bool isMatrixAFMT()
const {
return isImmTy(ImmTyMatrixAFMT); }
384 bool isMatrixBFMT()
const {
return isImmTy(ImmTyMatrixBFMT); }
385 bool isMatrixAScale()
const {
return isImmTy(ImmTyMatrixAScale); }
386 bool isMatrixBScale()
const {
return isImmTy(ImmTyMatrixBScale); }
387 bool isMatrixAScaleFmt()
const {
return isImmTy(ImmTyMatrixAScaleFmt); }
388 bool isMatrixBScaleFmt()
const {
return isImmTy(ImmTyMatrixBScaleFmt); }
389 bool isTFE()
const {
return isImmTy(ImmTyTFE); }
390 bool isFORMAT()
const {
return isImmTy(ImmTyFORMAT) &&
isUInt<7>(
getImm()); }
391 bool isDppFI()
const {
return isImmTy(ImmTyDppFI); }
392 bool isSDWADstSel()
const {
return isImmTy(ImmTySDWADstSel); }
393 bool isSDWASrc0Sel()
const {
return isImmTy(ImmTySDWASrc0Sel); }
394 bool isSDWASrc1Sel()
const {
return isImmTy(ImmTySDWASrc1Sel); }
395 bool isSDWADstUnused()
const {
return isImmTy(ImmTySDWADstUnused); }
396 bool isInterpSlot()
const {
return isImmTy(ImmTyInterpSlot); }
397 bool isInterpAttr()
const {
return isImmTy(ImmTyInterpAttr); }
398 bool isInterpAttrChan()
const {
return isImmTy(ImmTyInterpAttrChan); }
399 bool isOpSel()
const {
return isImmTy(ImmTyOpSel); }
400 bool isOpSelHi()
const {
return isImmTy(ImmTyOpSelHi); }
401 bool isNegLo()
const {
return isImmTy(ImmTyNegLo); }
402 bool isNegHi()
const {
return isImmTy(ImmTyNegHi); }
403 bool isDone()
const {
return isImmTy(ImmTyDone); }
404 bool isRowEn()
const {
return isImmTy(ImmTyRowEn); }
406 bool isRegOrImm()
const {
return isReg() || isImm(); }
408 bool isRegClass(
unsigned RCID)
const;
411 bool isRegClassTarget(
unsigned TargetRCIdx)
const;
415 bool isRegOrInlineNoMods(
unsigned RCID, MVT type)
const {
416 return isRegOrInline(RCID, type) && !hasModifiers();
419 bool isRegOrInlineNoModsTarget(
unsigned TargetRCIdx, MVT type)
const {
420 return isRegOrInlineTarget(TargetRCIdx, type) && !hasModifiers();
423 bool isSCSrcB16()
const {
424 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i16);
427 bool isSCSrc_b32()
const {
428 return isRegOrInlineNoMods(AMDGPU::SReg_32RegClassID, MVT::i32);
431 bool isSCSrc_b64()
const {
432 return isRegOrInlineNoMods(AMDGPU::SReg_64RegClassID, MVT::i64);
435 bool isBoolReg()
const;
437 bool isSSrc_b32()
const {
438 return isSCSrc_b32() || isLiteralImm(MVT::i32) || isExpr();
441 bool isSSrc_b16()
const {
return isSCSrcB16() || isLiteralImm(MVT::i16); }
443 bool isSSrc_b64()
const {
446 return isSCSrc_b64() || isLiteralImm(MVT::i64) ||
447 (((
const MCTargetAsmParser *)AsmParser)
448 ->getAvailableFeatures()[AMDGPU::Feature64BitLiterals] &&
452 bool isSSrc_f32()
const {
453 return isSCSrc_b32() || isLiteralImm(MVT::f32) || isExpr();
456 bool isSSrc_bf16()
const {
return isSCSrcB16() || isLiteralImm(MVT::bf16); }
458 bool isSSrc_f16()
const {
return isSCSrcB16() || isLiteralImm(MVT::f16); }
460 bool isSSrc_NoInline_f16()
const {
return isSSrc_f16(); }
462 bool isSSrcOrLds_b32()
const {
463 return isRegOrInlineNoMods(AMDGPU::SRegOrLds_32RegClassID, MVT::i32) ||
464 isLiteralImm(MVT::i32) || isExpr();
467 bool isVCSrc_b32()
const {
468 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i32);
471 bool isVCSrc_b32_Lo256()
const {
472 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo256RegClassID, MVT::i32);
475 bool isVCSrc_b64_Lo256()
const {
476 return isRegOrInlineNoMods(AMDGPU::VS_64_Lo256RegClassID, MVT::i64);
479 bool isVCSrc_b64()
const {
480 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::i64);
483 bool isVCSrcT_b16()
const {
484 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::i16);
487 bool isVCSrcTB16_Lo128()
const {
488 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::i16);
491 bool isVCSrcFake16B16_Lo128()
const {
492 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::i16);
495 bool isVCSrc_b16()
const {
496 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::i16);
499 bool isVCSrc_v2b16()
const {
return isVCSrc_b16(); }
501 bool isVCSrc_f32()
const {
502 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f32);
505 bool isVCSrc_f64()
const {
506 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64);
509 bool isVCSrcTBF16()
const {
510 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::bf16);
513 bool isVCSrcT_f16()
const {
514 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
517 bool isVCSrcT_bf16()
const {
518 return isRegOrInlineNoMods(AMDGPU::VS_16RegClassID, MVT::f16);
521 bool isVCSrcTF16_LO16()
const {
522 return isRegOrInlineNoMods(AMDGPU::VS_16_LO16RegClassID, MVT::f16);
525 bool isVCSrcTBF16_Lo128()
const {
526 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::bf16);
529 bool isVCSrcTF16_Lo128()
const {
530 return isRegOrInlineNoMods(AMDGPU::VS_16_Lo128RegClassID, MVT::f16);
533 bool isVCSrcFake16BF16_Lo128()
const {
534 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::bf16);
537 bool isVCSrcFake16F16_Lo128()
const {
538 return isRegOrInlineNoMods(AMDGPU::VS_32_Lo128RegClassID, MVT::f16);
541 bool isVCSrc_bf16()
const {
542 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::bf16);
545 bool isVCSrc_f16()
const {
546 return isRegOrInlineNoMods(AMDGPU::VS_32RegClassID, MVT::f16);
549 bool isVCSrc_v2bf16()
const {
return isVCSrc_bf16(); }
551 bool isVCSrc_v2f16()
const {
return isVCSrc_f16(); }
553 bool isVSrc_b32()
const {
554 return isVCSrc_f32() || isLiteralImm(MVT::i32) || isExpr();
557 bool isVSrc_b64()
const {
return isVCSrc_f64() || isLiteralImm(MVT::i64); }
559 bool isVSrc_v2b64()
const {
560 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::i64) ||
561 isLiteralImm(MVT::i64);
564 bool isVSrc_v2f64()
const {
565 return isRegOrInlineNoMods(AMDGPU::VS_128RegClassID, MVT::f64) ||
566 isLiteralImm(MVT::f64);
569 bool isVSrcT_b16()
const {
return isVCSrcT_b16() || isLiteralImm(MVT::i16); }
571 bool isVSrcT_b16_Lo128()
const {
572 return isVCSrcTB16_Lo128() || isLiteralImm(MVT::i16);
575 bool isVSrcFake16_b16_Lo128()
const {
576 return isVCSrcFake16B16_Lo128() || isLiteralImm(MVT::i16);
579 bool isVSrc_b16()
const {
return isVCSrc_b16() || isLiteralImm(MVT::i16); }
581 bool isVSrc_v2b16()
const {
return isVSrc_b16() || isLiteralImm(MVT::v2i16); }
583 bool isVSrc_v2f32()
const {
return isVSrc_f64() || isLiteralImm(MVT::v2f32); }
585 bool isVCSrc_v2b32()
const {
return isVCSrc_b64(); }
587 bool isVSrc_v2b32()
const {
return isVSrc_b64() || isLiteralImm(MVT::v2i32); }
589 bool isVSrc_f32()
const {
590 return isVCSrc_f32() || isLiteralImm(MVT::f32) || isExpr();
593 bool isVSrc_f64()
const {
594 return isRegOrInlineNoModsTarget(AMDGPU::VS_64_AlignTarget, MVT::f64) ||
595 isLiteralImm(MVT::f64);
598 bool isVSrcT_bf16()
const {
599 return isVCSrcTBF16() || isLiteralImm(MVT::bf16);
602 bool isVSrcT_f16()
const {
return isVCSrcT_f16() || isLiteralImm(MVT::f16); }
604 bool isVSrcT_f16_LO16()
const {
605 return isVCSrcTF16_LO16() || isLiteralImm(MVT::f16);
608 bool isVSrcT_bf16_Lo128()
const {
609 return isVCSrcTBF16_Lo128() || isLiteralImm(MVT::bf16);
612 bool isVSrcT_f16_Lo128()
const {
613 return isVCSrcTF16_Lo128() || isLiteralImm(MVT::f16);
616 bool isVSrcFake16_bf16_Lo128()
const {
617 return isVCSrcFake16BF16_Lo128() || isLiteralImm(MVT::bf16);
620 bool isVSrcFake16_f16_Lo128()
const {
621 return isVCSrcFake16F16_Lo128() || isLiteralImm(MVT::f16);
624 bool isVSrc_bf16()
const {
return isVCSrc_bf16() || isLiteralImm(MVT::bf16); }
626 bool isVSrc_f16()
const {
return isVCSrc_f16() || isLiteralImm(MVT::f16); }
628 bool isVSrc_v2bf16()
const {
629 return isVSrc_bf16() || isLiteralImm(MVT::v2bf16);
632 bool isVSrc_v2f16()
const {
return isVSrc_f16() || isLiteralImm(MVT::v2f16); }
634 bool isVSrc_v2f16_splat()
const {
return isVSrc_v2f16(); }
636 bool isVSrc_NoInline_v2f16()
const {
return isVSrc_v2f16(); }
638 bool isVISrc_64_bf16()
const {
639 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::bf16);
642 bool isVISrc_64_f16()
const {
643 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f16);
646 bool isVISrc_64_b32()
const {
647 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::i32);
650 bool isVISrc_64_f64()
const {
651 return isRegOrInlineNoModsTarget(AMDGPU::VReg_64_AlignTarget, MVT::f64);
654 bool isVISrc_256_b32()
const {
655 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::i32);
658 bool isVISrc_256_f32()
const {
659 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f32);
662 bool isVISrc_256_f64()
const {
663 return isRegOrInlineNoModsTarget(AMDGPU::VReg_256_AlignTarget, MVT::f64);
666 bool isVISrc_512_f64()
const {
667 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f64);
670 bool isVISrc_128_b32()
const {
671 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::i32);
674 bool isVISrc_128_f32()
const {
675 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f32);
678 bool isVISrc_512_b32()
const {
679 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::i32);
682 bool isVISrc_512_f32()
const {
683 return isRegOrInlineNoModsTarget(AMDGPU::VReg_512_AlignTarget, MVT::f32);
686 bool isVISrc_1024_b32()
const {
687 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::i32);
690 bool isVISrc_1024_f32()
const {
691 return isRegOrInlineNoModsTarget(AMDGPU::VReg_1024_AlignTarget, MVT::f32);
694 bool isAISrc_64_f64()
const {
695 return isRegOrInlineNoModsTarget(AMDGPU::AReg_64_AlignTarget, MVT::f64);
698 bool isAISrc_128_b32()
const {
699 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::i32);
702 bool isAISrc_128_f32()
const {
703 return isRegOrInlineNoModsTarget(AMDGPU::AReg_128_AlignTarget, MVT::f32);
706 bool isVISrc_128_bf16()
const {
707 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::bf16);
710 bool isVISrc_128_f16()
const {
711 return isRegOrInlineNoModsTarget(AMDGPU::VReg_128_AlignTarget, MVT::f16);
714 bool isAISrc_256_f64()
const {
715 return isRegOrInlineNoModsTarget(AMDGPU::AReg_256_AlignTarget, MVT::f64);
718 bool isAISrc_512_b32()
const {
719 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::i32);
722 bool isAISrc_512_f32()
const {
723 return isRegOrInlineNoModsTarget(AMDGPU::AReg_512_AlignTarget, MVT::f32);
726 bool isAISrc_1024_b32()
const {
727 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::i32);
730 bool isAISrc_1024_f32()
const {
731 return isRegOrInlineNoModsTarget(AMDGPU::AReg_1024_AlignTarget, MVT::f32);
734 bool isKImmFP32()
const {
return isLiteralImm(MVT::f32); }
736 bool isKImmFP16()
const {
return isLiteralImm(MVT::f16); }
738 bool isKImmFP64()
const {
return isLiteralImm(MVT::f64); }
740 bool isMem()
const override {
return false; }
742 bool isExpr()
const {
return Kind == Expression; }
744 bool isSOPPBrTarget()
const {
return isExpr() || isImm(); }
746 bool isSWaitCnt()
const;
747 bool isDepCtr()
const;
748 bool isSDelayALU()
const;
749 bool isHwreg()
const;
750 bool isSendMsg()
const;
751 bool isWaitEvent()
const;
752 bool isSplitBarrier()
const;
753 bool isSwizzle()
const;
754 bool isSMRDOffset8()
const;
755 bool isSMEMOffset()
const;
756 bool isSMRDLiteralOffset()
const;
758 bool isDPPCtrl()
const;
760 bool isGPRIdxMode()
const;
761 bool isS16Imm()
const;
762 bool isU16Imm()
const;
763 bool isEndpgm()
const;
765 auto getPredicate(std::function<
bool(
const AMDGPUOperand &
Op)>
P)
const {
766 return [
this,
P]() {
return P(*
this); };
771 return StringRef(Tok.Data, Tok.Length);
779 void setImm(int64_t Val) {
784 ImmTy getImmTy()
const {
789 MCRegister
getReg()
const override {
794 SMLoc getStartLoc()
const override {
return StartLoc; }
796 SMLoc getEndLoc()
const override {
return EndLoc; }
798 int getMCOpIdx()
const {
return MCOpIdx; }
800 Modifiers getModifiers()
const {
801 assert(isRegKind() || isImmTy(ImmTyNone));
802 return isRegKind() ?
Reg.Mods :
Imm.Mods;
805 void setModifiers(Modifiers Mods) {
806 assert(isRegKind() || isImmTy(ImmTyNone));
813 bool hasModifiers()
const {
return getModifiers().hasModifiers(); }
815 bool hasFPModifiers()
const {
return getModifiers().hasFPModifiers(); }
817 bool hasIntModifiers()
const {
return getModifiers().hasIntModifiers(); }
819 bool isForcedLit()
const {
820 return isImmLiteral() && getModifiers().isForcedLit();
823 bool isForcedLit64()
const {
824 return isImmLiteral() && getModifiers().isForcedLit64();
829 void addImmOperands(MCInst &Inst,
unsigned N,
830 bool ApplyModifiers =
true)
const;
832 void addLiteralImmOperand(MCInst &Inst, int64_t Val,
833 bool ApplyModifiers)
const;
835 void addRegOperands(MCInst &Inst,
unsigned N)
const;
837 void addRegOrImmOperands(MCInst &Inst,
unsigned N)
const {
839 addRegOperands(Inst,
N);
841 addImmOperands(Inst,
N);
844 void addRegOrImmWithInputModsOperands(MCInst &Inst,
unsigned N)
const {
845 Modifiers Mods = getModifiers();
848 addRegOperands(Inst,
N);
850 addImmOperands(Inst,
N,
false);
854 void addRegOrImmWithFPInputModsOperands(MCInst &Inst,
unsigned N)
const {
855 assert(!hasIntModifiers());
856 addRegOrImmWithInputModsOperands(Inst,
N);
859 void addRegWithInputModsOperands(MCInst &Inst,
unsigned N)
const {
860 Modifiers Mods = getModifiers();
863 addRegOperands(Inst,
N);
866 void addRegWithFPInputModsOperands(MCInst &Inst,
unsigned N)
const {
867 assert(!hasIntModifiers());
868 addRegWithInputModsOperands(Inst,
N);
871 static void printImmTy(raw_ostream &OS, ImmTy
Type) {
874 case ImmTyNone: OS <<
"None";
break;
875 case ImmTyGDS: OS <<
"GDS";
break;
876 case ImmTyLDS: OS <<
"LDS";
break;
877 case ImmTyOffen: OS <<
"Offen";
break;
878 case ImmTyIdxen: OS <<
"Idxen";
break;
879 case ImmTyAddr64: OS <<
"Addr64";
break;
880 case ImmTyOffset: OS <<
"Offset";
break;
881 case ImmTyInstOffset: OS <<
"InstOffset";
break;
882 case ImmTyOffset0: OS <<
"Offset0";
break;
883 case ImmTyOffset1: OS <<
"Offset1";
break;
884 case ImmTySMEMOffsetMod: OS <<
"SMEMOffsetMod";
break;
885 case ImmTyCPol: OS <<
"CPol";
break;
886 case ImmTyIndexKey8bit: OS <<
"index_key";
break;
887 case ImmTyIndexKey16bit: OS <<
"index_key";
break;
888 case ImmTyIndexKey32bit: OS <<
"index_key";
break;
889 case ImmTyTFE: OS <<
"TFE";
break;
890 case ImmTyIsAsync: OS <<
"IsAsync";
break;
891 case ImmTyD16: OS <<
"D16";
break;
892 case ImmTyFORMAT: OS <<
"FORMAT";
break;
893 case ImmTyClamp: OS <<
"Clamp";
break;
894 case ImmTyOModSI: OS <<
"OModSI";
break;
895 case ImmTyDPP8: OS <<
"DPP8";
break;
896 case ImmTyDppCtrl: OS <<
"DppCtrl";
break;
897 case ImmTyDppRowMask: OS <<
"DppRowMask";
break;
898 case ImmTyDppBankMask: OS <<
"DppBankMask";
break;
899 case ImmTyDppBoundCtrl: OS <<
"DppBoundCtrl";
break;
900 case ImmTyDppFI: OS <<
"DppFI";
break;
901 case ImmTySDWADstSel: OS <<
"SDWADstSel";
break;
902 case ImmTySDWASrc0Sel: OS <<
"SDWASrc0Sel";
break;
903 case ImmTySDWASrc1Sel: OS <<
"SDWASrc1Sel";
break;
904 case ImmTySDWADstUnused: OS <<
"SDWADstUnused";
break;
905 case ImmTyDMask: OS <<
"DMask";
break;
906 case ImmTyDim: OS <<
"Dim";
break;
907 case ImmTyUNorm: OS <<
"UNorm";
break;
908 case ImmTyDA: OS <<
"DA";
break;
909 case ImmTyR128A16: OS <<
"R128A16";
break;
910 case ImmTyA16: OS <<
"A16";
break;
911 case ImmTyLWE: OS <<
"LWE";
break;
912 case ImmTyOff: OS <<
"Off";
break;
913 case ImmTyExpTgt: OS <<
"ExpTgt";
break;
914 case ImmTyExpCompr: OS <<
"ExpCompr";
break;
915 case ImmTyExpVM: OS <<
"ExpVM";
break;
916 case ImmTyDone: OS <<
"Done";
break;
917 case ImmTyRowEn: OS <<
"RowEn";
break;
918 case ImmTyHwreg: OS <<
"Hwreg";
break;
919 case ImmTySendMsg: OS <<
"SendMsg";
break;
920 case ImmTyWaitEvent: OS <<
"WaitEvent";
break;
921 case ImmTyInterpSlot: OS <<
"InterpSlot";
break;
922 case ImmTyInterpAttr: OS <<
"InterpAttr";
break;
923 case ImmTyInterpAttrChan: OS <<
"InterpAttrChan";
break;
924 case ImmTyOpSel: OS <<
"OpSel";
break;
925 case ImmTyOpSelHi: OS <<
"OpSelHi";
break;
926 case ImmTyNegLo: OS <<
"NegLo";
break;
927 case ImmTyNegHi: OS <<
"NegHi";
break;
928 case ImmTySwizzle: OS <<
"Swizzle";
break;
929 case ImmTyGprIdxMode: OS <<
"GprIdxMode";
break;
930 case ImmTyHigh: OS <<
"High";
break;
931 case ImmTyBLGP: OS <<
"BLGP";
break;
932 case ImmTyCBSZ: OS <<
"CBSZ";
break;
933 case ImmTyABID: OS <<
"ABID";
break;
934 case ImmTyEndpgm: OS <<
"Endpgm";
break;
935 case ImmTyWaitVDST: OS <<
"WaitVDST";
break;
936 case ImmTyWaitEXP: OS <<
"WaitEXP";
break;
937 case ImmTyWaitVAVDst: OS <<
"WaitVAVDst";
break;
938 case ImmTyWaitVMVSrc: OS <<
"WaitVMVSrc";
break;
939 case ImmTyBitOp3: OS <<
"BitOp3";
break;
940 case ImmTyMatrixAFMT: OS <<
"ImmTyMatrixAFMT";
break;
941 case ImmTyMatrixBFMT: OS <<
"ImmTyMatrixBFMT";
break;
942 case ImmTyMatrixAScale: OS <<
"ImmTyMatrixAScale";
break;
943 case ImmTyMatrixBScale: OS <<
"ImmTyMatrixBScale";
break;
944 case ImmTyMatrixAScaleFmt: OS <<
"ImmTyMatrixAScaleFmt";
break;
945 case ImmTyMatrixBScaleFmt: OS <<
"ImmTyMatrixBScaleFmt";
break;
946 case ImmTyMatrixAReuse: OS <<
"ImmTyMatrixAReuse";
break;
947 case ImmTyMatrixBReuse: OS <<
"ImmTyMatrixBReuse";
break;
948 case ImmTyScaleSel: OS <<
"ScaleSel" ;
break;
949 case ImmTyByteSel: OS <<
"ByteSel" ;
break;
954 void print(raw_ostream &OS,
const MCAsmInfo &MAI)
const override {
958 <<
" mods: " <<
Reg.Mods <<
'>';
962 if (getImmTy() != ImmTyNone) {
964 printImmTy(OS, getImmTy());
966 OS <<
" mods: " <<
Imm.Mods <<
'>';
979 static AMDGPUOperand::Ptr CreateImm(
const AMDGPUAsmParser *AsmParser,
980 int64_t Val, SMLoc Loc,
981 ImmTy
Type = ImmTyNone,
982 bool IsFPImm =
false) {
983 auto Op = std::make_unique<AMDGPUOperand>(Immediate, AsmParser);
985 Op->Imm.IsFPImm = IsFPImm;
987 Op->Imm.Mods = Modifiers();
993 static AMDGPUOperand::Ptr CreateToken(
const AMDGPUAsmParser *AsmParser,
994 StringRef Str, SMLoc Loc,
995 bool HasExplicitEncodingSize =
true) {
996 auto Res = std::make_unique<AMDGPUOperand>(Token, AsmParser);
997 Res->Tok.Data = Str.data();
998 Res->Tok.Length = Str.size();
1004 static AMDGPUOperand::Ptr CreateReg(
const AMDGPUAsmParser *AsmParser,
1005 MCRegister
Reg, SMLoc S, SMLoc
E) {
1006 auto Op = std::make_unique<AMDGPUOperand>(Register, AsmParser);
1007 Op->Reg.RegNo =
Reg;
1008 Op->Reg.Mods = Modifiers();
1014 static AMDGPUOperand::Ptr CreateExpr(
const AMDGPUAsmParser *AsmParser,
1015 const class MCExpr *Expr, SMLoc S) {
1016 auto Op = std::make_unique<AMDGPUOperand>(Expression, AsmParser);
1025 OS <<
"abs:" << Mods.Abs <<
" neg: " << Mods.Neg <<
" sext:" << Mods.Sext;
1034#define GET_REGISTER_MATCHER
1035#include "AMDGPUGenAsmMatcher.inc"
1036#undef GET_REGISTER_MATCHER
1037#undef GET_SUBTARGET_FEATURE_NAME
1042class KernelScopeInfo {
1043 int SgprIndexUnusedMin = -1;
1044 int VgprIndexUnusedMin = -1;
1045 int AgprIndexUnusedMin = -1;
1049 void usesSgprAt(
int i) {
1050 if (i >= SgprIndexUnusedMin) {
1051 SgprIndexUnusedMin = ++i;
1054 Ctx->getOrCreateSymbol(
Twine(
".kernel.sgpr_count"));
1060 void usesVgprAt(
int i) {
1061 if (i >= VgprIndexUnusedMin) {
1062 VgprIndexUnusedMin = ++i;
1065 Ctx->getOrCreateSymbol(
Twine(
".kernel.vgpr_count"));
1067 VgprIndexUnusedMin);
1073 void usesAgprAt(
int i) {
1078 if (i >= AgprIndexUnusedMin) {
1079 AgprIndexUnusedMin = ++i;
1082 Ctx->getOrCreateSymbol(
Twine(
".kernel.agpr_count"));
1087 Ctx->getOrCreateSymbol(
Twine(
".kernel.vgpr_count"));
1089 VgprIndexUnusedMin);
1096 KernelScopeInfo() =
default;
1100 MSTI = Ctx->getSubtargetInfo();
1102 usesSgprAt(SgprIndexUnusedMin = -1);
1103 usesVgprAt(VgprIndexUnusedMin = -1);
1105 usesAgprAt(AgprIndexUnusedMin = -1);
1109 void usesRegister(RegisterKind RegKind,
unsigned DwordRegIndex,
1110 unsigned RegWidth) {
1113 usesSgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1116 usesAgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1119 usesVgprAt(DwordRegIndex +
divideCeil(RegWidth, 32) - 1);
1128 MCAsmParser &Parser;
1130 unsigned ForcedEncodingSize = 0;
1131 bool ForcedDPP =
false;
1132 bool ForcedSDWA =
false;
1133 KernelScopeInfo KernelScope;
1134 const unsigned HwMode;
1136 const AMDGPU::IsaVersion ISA;
1141#define GET_ASSEMBLER_HEADER
1142#include "AMDGPUGenAsmMatcher.inc"
1147 unsigned getRegOperandSize(
const MCInstrDesc &
Desc,
unsigned OpNo)
const {
1149 int16_t RCID = MII.getOpRegClassID(
Desc.operands()[OpNo], HwMode);
1153 std::optional<AMDGPU::InfoSectionData> InfoData;
1160 bool TargetDirectiveEmitted =
false;
1169 SmallVector<unsigned> OpcodeStream;
1171 OpcodeStreamSymbols;
1172 SmallPtrSet<const MCSymbol *, 8> AMDHSAKernelSymbols;
1175 void checkKernelPrologues();
1178 void createConstantSymbol(StringRef Id, int64_t Val);
1180 bool ParseAsAbsoluteExpression(uint32_t &Ret);
1181 bool OutOfRangeError(SMRange
Range);
1197 bool calculateGPRBlocks(
const FeatureBitset &Features,
const MCExpr *VCCUsed,
1198 const MCExpr *FlatScrUsed,
bool XNACKUsed,
1199 std::optional<bool> EnableWavefrontSize32,
1200 const MCExpr *NextFreeVGPR, SMRange VGPRRange,
1201 const MCExpr *NextFreeSGPR, SMRange SGPRRange,
1202 const MCExpr *&VGPRBlocks,
const MCExpr *&SGPRBlocks);
1203 bool ParseDirectiveAMDGCNTarget();
1204 bool ParseDirectiveAMDHSACodeObjectVersion();
1205 bool ParseDirectiveAMDHSAKernel();
1206 bool ParseAMDKernelCodeTValue(StringRef ID, AMDGPUMCKernelCodeT &Header);
1207 bool ParseDirectiveAMDKernelCodeT();
1209 bool subtargetHasRegister(
const MCRegisterInfo &MRI, MCRegister
Reg);
1210 bool ParseDirectiveAMDGPUHsaKernel();
1212 bool ParseDirectiveISAVersion();
1213 bool ParseDirectiveHSAMetadata();
1214 bool ParseDirectivePALMetadataBegin();
1215 bool ParseDirectivePALMetadata();
1216 bool ParseDirectiveAMDGPULDS();
1217 bool ParseDirectiveAMDGPUInfo();
1221 bool ParseToEndDirective(
const char *AssemblerDirectiveBegin,
1222 const char *AssemblerDirectiveEnd,
1223 std::string &CollectString);
1225 bool AddNextRegisterToList(MCRegister &
Reg,
unsigned &RegWidth,
1226 RegisterKind RegKind, MCRegister Reg1,
1227 RegisterKind RegKind1, SMLoc Loc);
1228 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &
Reg,
1229 unsigned &RegNum,
unsigned &RegWidth,
1230 bool RestoreOnFailure =
false);
1231 bool ParseAMDGPURegister(RegisterKind &RegKind, MCRegister &
Reg,
1232 unsigned &RegNum,
unsigned &RegWidth,
1233 SmallVectorImpl<AsmToken> &Tokens);
1234 MCRegister ParseRegularReg(RegisterKind &RegKind,
unsigned &RegNum,
1236 SmallVectorImpl<AsmToken> &Tokens);
1237 MCRegister ParseSpecialReg(RegisterKind &RegKind,
unsigned &RegNum,
1239 SmallVectorImpl<AsmToken> &Tokens);
1240 MCRegister ParseRegList(RegisterKind &RegKind,
unsigned &RegNum,
1242 SmallVectorImpl<AsmToken> &Tokens);
1243 bool ParseRegRange(
unsigned &Num,
unsigned &Width,
unsigned &SubReg);
1244 MCRegister getRegularReg(RegisterKind RegKind,
unsigned RegNum,
1245 unsigned SubReg,
unsigned RegWidth, SMLoc Loc);
1248 bool isRegister(
const AsmToken &Token,
const AsmToken &NextToken)
const;
1249 std::optional<StringRef> getGprCountSymbolName(RegisterKind RegKind);
1250 void initializeGprCountSymbol(RegisterKind RegKind);
1251 bool updateGprCountSymbols(RegisterKind RegKind,
unsigned DwordRegIndex,
1257 OperandMode_Default,
1261 using OptionalImmIndexMap = std::map<AMDGPUOperand::ImmTy, unsigned>;
1263 AMDGPUAsmParser(
const MCSubtargetInfo &STI, MCAsmParser &_Parser,
1264 const MCInstrInfo &MII)
1265 : MCTargetAsmParser(STI, MII), Parser(_Parser),
1266 HwMode(STI.getHwMode(MCSubtargetInfo::HwMode_RegInfo)),
1271 setAvailableFeatures(ComputeAvailableFeatures(getFeatureBits()));
1273 if (ISA.Major >= 6 &&
isHsaAbi(getSTI())) {
1274 createConstantSymbol(
".amdgcn.gfx_generation_number", ISA.Major);
1275 createConstantSymbol(
".amdgcn.gfx_generation_minor", ISA.Minor);
1276 createConstantSymbol(
".amdgcn.gfx_generation_stepping", ISA.Stepping);
1278 createConstantSymbol(
".option.machine_version_major", ISA.Major);
1279 createConstantSymbol(
".option.machine_version_minor", ISA.Minor);
1280 createConstantSymbol(
".option.machine_version_stepping", ISA.Stepping);
1282 if (ISA.Major >= 6 &&
isHsaAbi(getSTI())) {
1283 initializeGprCountSymbol(IS_VGPR);
1284 initializeGprCountSymbol(IS_SGPR);
1289 createConstantSymbol(Symbol, Code);
1291 createConstantSymbol(
"UC_VERSION_W64_BIT", 0x2000);
1292 createConstantSymbol(
"UC_VERSION_W32_BIT", 0x4000);
1293 createConstantSymbol(
"UC_VERSION_MDP_BIT", 0x8000);
1333 bool hasBVHRayTracingInsts()
const {
1334 return getFeatureBits()[AMDGPU::FeatureBVHRayTracingInsts];
1337 bool isWave32()
const {
return getAvailableFeatures()[Feature_isWave32Bit]; }
1339 bool isWave64()
const {
return getAvailableFeatures()[Feature_isWave64Bit]; }
1341 bool hasInv2PiInlineImm()
const {
1342 return getFeatureBits()[AMDGPU::FeatureInv2PiInlineImm];
1345 bool has64BitLiterals()
const {
1346 return getFeatureBits()[AMDGPU::Feature64BitLiterals];
1349 bool hasFlatOffsets()
const {
1350 return getFeatureBits()[AMDGPU::FeatureFlatInstOffsets];
1353 bool hasTrue16Insts()
const {
1354 return getFeatureBits()[AMDGPU::FeatureTrue16BitInsts];
1358 return getFeatureBits()[AMDGPU::FeatureArchitectedFlatScratch];
1361 bool hasSGPR102_SGPR103()
const {
return !
isVI() && !
isGFX9(); }
1363 bool hasSGPR104_SGPR105()
const {
return isGFX10Plus(); }
1365 bool hasIntClamp()
const {
return getFeatureBits()[AMDGPU::FeatureIntClamp]; }
1367 bool hasPartialNSAEncoding()
const {
1368 return getFeatureBits()[AMDGPU::FeaturePartialNSAEncoding];
1371 bool hasGloballyAddressableScratch()
const {
1372 return getFeatureBits()[AMDGPU::FeatureGloballyAddressableScratch];
1385 AMDGPUTargetStreamer &getTargetStreamer() {
1386 MCTargetStreamer &TS = *getParser().getStreamer().getTargetStreamer();
1387 return static_cast<AMDGPUTargetStreamer &
>(TS);
1393 return const_cast<AMDGPUAsmParser *
>(
this)->MCTargetAsmParser::getContext();
1396 const MCRegisterInfo *getMRI()
const {
1400 const MCInstrInfo *getMII()
const {
return &MII; }
1404 int16_t getTargetRegClass(
unsigned TargetRCIdx)
const {
1410 const FeatureBitset &getFeatureBits()
const {
1411 return getSTI().getFeatureBits();
1414 void setForcedEncodingSize(
unsigned Size) { ForcedEncodingSize =
Size; }
1415 void setForcedDPP(
bool ForceDPP_) { ForcedDPP = ForceDPP_; }
1416 void setForcedSDWA(
bool ForceSDWA_) { ForcedSDWA = ForceSDWA_; }
1418 unsigned getForcedEncodingSize()
const {
return ForcedEncodingSize; }
1419 bool isForcedVOP3()
const {
return ForcedEncodingSize == 64; }
1420 bool isForcedDPP()
const {
return ForcedDPP; }
1421 bool isForcedSDWA()
const {
return ForcedSDWA; }
1422 ArrayRef<unsigned> getMatchedVariants()
const;
1423 StringRef getMatchedVariantName()
const;
1425 std::unique_ptr<AMDGPUOperand> parseRegister(
bool RestoreOnFailure =
false);
1426 bool ParseRegister(MCRegister &RegNo, SMLoc &StartLoc, SMLoc &EndLoc,
1427 bool RestoreOnFailure);
1428 bool parseRegister(MCRegister &
Reg, SMLoc &StartLoc, SMLoc &EndLoc)
override;
1429 ParseStatus tryParseRegister(MCRegister &
Reg, SMLoc &StartLoc,
1430 SMLoc &EndLoc)
override;
1431 unsigned checkTargetMatchPredicate(MCInst &Inst)
override;
1432 unsigned validateTargetOperandClass(MCParsedAsmOperand &
Op,
1433 unsigned Kind)
override;
1434 bool matchAndEmitInstruction(SMLoc IDLoc,
unsigned &Opcode,
1437 bool MatchingInlineAsm)
override;
1438 bool ParseDirective(AsmToken DirectiveID)
override;
1439 void doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc)
override;
1440 void onEndOfFile()
override;
1442 OperandMode
Mode = OperandMode_Default);
1443 StringRef parseMnemonicSuffix(StringRef Name);
1444 bool parseInstruction(ParseInstructionInfo &Info, StringRef Name,
1450 ParseStatus parseIntWithPrefix(
const char *Prefix, int64_t &
Int);
1454 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1455 std::function<
bool(int64_t &)> ConvertResult =
nullptr);
1457 ParseStatus parseOperandArrayWithPrefix(
1459 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1460 bool (*ConvertResult)(int64_t &) =
nullptr);
1464 AMDGPUOperand::ImmTy ImmTy = AMDGPUOperand::ImmTyNone,
1465 bool IgnoreNegative =
false);
1466 unsigned getCPolKind(StringRef Id, StringRef Mnemo,
bool &Disabling)
const;
1470 ParseStatus parseStringWithPrefix(StringRef Prefix, StringRef &
Value,
1474 ArrayRef<const char *> Ids,
1478 ArrayRef<const char *> Ids,
1479 AMDGPUOperand::ImmTy
Type);
1482 bool isOperandModifier(
const AsmToken &Token,
1483 const AsmToken &NextToken)
const;
1484 bool isRegOrOperandModifier(
const AsmToken &Token,
1485 const AsmToken &NextToken)
const;
1486 bool isNamedOperandModifier(
const AsmToken &Token,
1487 const AsmToken &NextToken)
const;
1488 bool isOpcodeModifierWithVal(
const AsmToken &Token,
1489 const AsmToken &NextToken)
const;
1490 bool parseSP3NegModifier();
1497 bool AllowImm =
true);
1499 bool AllowImm =
true);
1505 AMDGPUOperand::ImmTy ImmTy);
1510 AMDGPUOperand::ImmTy
Type);
1514 AMDGPUOperand::ImmTy
Type);
1518 AMDGPUOperand::ImmTy
Type);
1522 ParseStatus parseDfmtNfmt(int64_t &
Format);
1523 ParseStatus parseUfmt(int64_t &
Format);
1524 ParseStatus parseSymbolicSplitFormat(StringRef FormatStr, SMLoc Loc,
1526 ParseStatus parseSymbolicUnifiedFormat(StringRef FormatStr, SMLoc Loc,
1529 ParseStatus parseSymbolicOrNumericFormat(int64_t &
Format);
1530 ParseStatus parseNumericFormat(int64_t &
Format);
1534 bool tryParseFmt(
const char *Pref, int64_t MaxVal, int64_t &Val);
1535 bool matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt, StringRef FormatStr,
1540 bool parseCnt(int64_t &IntVal);
1543 bool parseDepCtr(int64_t &IntVal,
unsigned &Mask);
1544 void depCtrError(SMLoc Loc,
int ErrorId, StringRef DepCtrName);
1547 bool parseDelay(int64_t &Delay);
1553 struct OperandInfoTy {
1556 bool IsSymbolic =
false;
1557 bool IsDefined =
false;
1559 constexpr OperandInfoTy(int64_t Val) : Val(Val) {}
1562 struct StructuredOpField : OperandInfoTy {
1566 bool IsDefined =
false;
1568 constexpr StructuredOpField(StringLiteral Id, StringLiteral Desc,
1569 unsigned Width, int64_t
Default)
1570 : OperandInfoTy(
Default), Id(Id), Desc(Desc), Width(Width) {}
1571 virtual ~StructuredOpField() =
default;
1573 bool Error(AMDGPUAsmParser &Parser,
const Twine &Err)
const {
1574 Parser.Error(Loc,
"invalid " + Desc +
": " + Err);
1578 virtual bool validate(AMDGPUAsmParser &Parser)
const {
1580 return Error(Parser,
"not supported on this GPU");
1582 return Error(Parser,
"only " + Twine(Width) +
"-bit values are legal");
1590 bool parseSendMsgBody(OperandInfoTy &
Msg, OperandInfoTy &
Op,
1591 OperandInfoTy &Stream);
1592 bool validateSendMsg(
const OperandInfoTy &
Msg,
const OperandInfoTy &
Op,
1593 const OperandInfoTy &Stream);
1595 ParseStatus parseHwregFunc(OperandInfoTy &HwReg, OperandInfoTy &
Offset,
1596 OperandInfoTy &Width);
1601 static SMLoc getLaterLoc(SMLoc a, SMLoc b);
1608 SMLoc getOperandLoc(std::function<
bool(
const AMDGPUOperand &)>
Test,
1610 SMLoc getImmLoc(AMDGPUOperand::ImmTy
Type,
1614 bool validateInstruction(
const MCInst &Inst, SMLoc IDLoc,
1619 bool validateBF16InlineConst(
const MCInst &Inst,
1622 bool validateConstantBusLimitations(
const MCInst &Inst,
1624 std::optional<unsigned> checkVOPDRegBankConstraints(
const MCInst &Inst,
1627 bool tryVOPD(
const MCInst &Inst);
1628 bool tryVOPD3(
const MCInst &Inst);
1629 bool tryAnotherVOPDEncoding(
const MCInst &Inst);
1631 bool validateIntClampSupported(
const MCInst &Inst);
1632 bool validateMIMGAtomicDMask(
const MCInst &Inst);
1633 bool validateMIMGGatherDMask(
const MCInst &Inst);
1635 bool validateMIMGDataSize(
const MCInst &Inst, SMLoc IDLoc);
1636 bool validateMIMGAddrSize(
const MCInst &Inst, SMLoc IDLoc);
1637 bool validateMIMGD16(
const MCInst &Inst);
1639 bool validateTensorR128(
const MCInst &Inst);
1640 bool validateMIMGMSAA(
const MCInst &Inst);
1641 bool validateOpSel(
const MCInst &Inst);
1642 bool validateTrue16OpSel(
const MCInst &Inst);
1643 bool validateNeg(
const MCInst &Inst, AMDGPU::OpName OpName);
1645 bool validateVccOperand(MCRegister
Reg)
const;
1650 bool validateAGPRLdSt(
const MCInst &Inst)
const;
1651 bool validateVGPRAlign(
const MCInst &Inst)
const;
1655 bool validateDivScale(
const MCInst &Inst);
1660 const unsigned CPol);
1665 bool validateClusterBarrierIsFirst(
const MCInst &Inst,
1668 unsigned getConstantBusLimit(
unsigned Opcode)
const;
1669 bool usesConstantBus(
const MCInst &Inst,
unsigned OpIdx);
1670 bool isInlineConstant(
const MCInst &Inst,
unsigned OpIdx)
const;
1671 MCRegister findImplicitSGPRReadInVOP(
const MCInst &Inst)
const;
1673 bool isSupportedMnemo(StringRef Mnemo,
const FeatureBitset &FBS);
1674 bool isSupportedMnemo(StringRef Mnemo,
const FeatureBitset &FBS,
1675 ArrayRef<unsigned> Variants);
1676 bool checkUnsupportedInstruction(StringRef Name, SMLoc IDLoc);
1678 bool isId(
const StringRef Id)
const;
1679 bool isId(
const AsmToken &Token,
const StringRef Id)
const;
1681 StringRef getId()
const;
1682 bool trySkipId(
const StringRef Id);
1683 bool trySkipId(
const StringRef Pref,
const StringRef Id);
1687 bool parseString(StringRef &Val,
1688 const StringRef ErrMsg =
"expected a string");
1689 bool parseId(StringRef &Val,
const StringRef ErrMsg =
"");
1695 StringRef getTokenStr()
const;
1696 AsmToken peekToken(
bool ShouldSkipSpace =
true);
1698 SMLoc getLoc()
const;
1702 void onBeginOfFile()
override;
1706 void emitTargetDirective();
1707 bool parsePrimaryExpr(
const MCExpr *&Res, SMLoc &EndLoc)
override;
1719 bool parseSwizzleOperand(int64_t &
Op,
const unsigned MinVal,
1720 const unsigned MaxVal,
const Twine &ErrMsg,
1722 bool parseSwizzleOperands(
const unsigned OpNum, int64_t *
Op,
1723 const unsigned MinVal,
const unsigned MaxVal,
1724 const StringRef ErrMsg);
1726 bool parseSwizzleOffset(int64_t &
Imm);
1727 bool parseSwizzleMacro(int64_t &
Imm);
1728 bool parseSwizzleQuadPerm(int64_t &
Imm);
1729 bool parseSwizzleBitmaskPerm(int64_t &
Imm);
1730 bool parseSwizzleBroadcast(int64_t &
Imm);
1731 bool parseSwizzleSwap(int64_t &
Imm);
1732 bool parseSwizzleReverse(int64_t &
Imm);
1733 bool parseSwizzleFFT(int64_t &
Imm);
1734 bool parseSwizzleRotate(int64_t &
Imm);
1737 int64_t parseGPRIdxMacro();
1740 cvtMubufImpl(Inst,
Operands,
false);
1743 cvtMubufImpl(Inst,
Operands,
true);
1749 OptionalImmIndexMap &OptionalIdx);
1758 OptionalImmIndexMap &OptionalIdx);
1760 OptionalImmIndexMap &OptionalIdx);
1764 void cvtOpSelHelper(MCInst &Inst,
unsigned OpSel);
1766 bool parseDimId(
unsigned &Encoding);
1768 bool convertDppBoundCtrl(int64_t &BoundCtrl);
1772 int64_t parseDPPCtrlSel(StringRef Ctrl);
1773 int64_t parseDPPCtrlPerm();
1779 bool IsDPP8 =
false);
1785 AMDGPUOperand::ImmTy
Type);
1793 enum class SDWAInstType :
unsigned {
VOP1 = 0,
VOP2 = 1,
VOPC = 2 };
1796 SDWAInstType BasicInstType,
bool SkipDstVcc =
false,
1797 bool SkipSrcVcc =
false);
1907bool AMDGPUOperand::isInlinableImm(
MVT type)
const {
1917 if (!isImmTy(ImmTyNone)) {
1922 if (getModifiers().
Lit != LitModifier::None)
1932 if (type == MVT::f64 || type == MVT::i64) {
1934 AsmParser->hasInv2PiInlineImm());
1937 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64,
Imm.Val));
1956 APFloat::rmNearestTiesToEven, &Lost);
1963 uint32_t ImmVal = FPLiteral.bitcastToAPInt().getZExtValue();
1965 AsmParser->hasInv2PiInlineImm());
1970 static_cast<int32_t
>(FPLiteral.bitcastToAPInt().getZExtValue()),
1971 AsmParser->hasInv2PiInlineImm());
1975 if (type == MVT::f64 || type == MVT::i64) {
1977 AsmParser->hasInv2PiInlineImm());
1986 static_cast<int16_t
>(
Literal.getLoBits(16).getSExtValue()), type,
1987 AsmParser->hasInv2PiInlineImm());
1991 static_cast<int32_t
>(
Literal.getLoBits(32).getZExtValue()),
1992 AsmParser->hasInv2PiInlineImm());
1995bool AMDGPUOperand::isLiteralImm(MVT type)
const {
1997 if (!isImmTy(ImmTyNone)) {
2002 (type == MVT::i64 || type == MVT::f64) && AsmParser->has64BitLiterals();
2007 if (type == MVT::f64 && hasFPModifiers()) {
2027 if (type == MVT::f64) {
2032 if (type == MVT::i64) {
2045 MVT ExpectedType = (type == MVT::v2f16) ? MVT::f16
2046 : (type == MVT::v2i16) ? MVT::f32
2047 : (type == MVT::v2f32) ? MVT::f32
2050 APFloat FPLiteral(APFloat::IEEEdouble(), APInt(64,
Imm.Val));
2054bool AMDGPUOperand::isRegClassTarget(
unsigned TargetRCIdx)
const {
2057 int16_t RCID = AsmParser->getTargetRegClass(TargetRCIdx);
2058 return RCID >= 0 && isRegClass(RCID);
2061bool AMDGPUOperand::isRegClass(
unsigned RCID)
const {
2062 return isRegKind() &&
2063 AsmParser->getMRI()->getRegClass(RCID).contains(
getReg());
2066bool AMDGPUOperand::isVRegWithInputMods()
const {
2067 return isRegClass(AMDGPU::VGPR_32RegClassID) ||
2069 (AsmParser->getFeatureBits()[AMDGPU::FeatureDPALU_DPP] &&
2070 isRegClassTarget(AMDGPU::VReg_64_AlignTarget));
2073template <
bool IsFake16>
2074bool AMDGPUOperand::isT16_Lo128VRegWithInputMods()
const {
2075 return isRegClass(IsFake16 ? AMDGPU::VGPR_32_Lo128RegClassID
2076 : AMDGPU::VGPR_16_Lo128RegClassID);
2079template <
bool IsFake16>
bool AMDGPUOperand::isT16VRegWithInputMods()
const {
2080 return isRegClass(IsFake16 ? AMDGPU::VGPR_32RegClassID
2081 : AMDGPU::VGPR_16RegClassID);
2084bool AMDGPUOperand::isT16_LO16VRegWithInputMods()
const {
2085 return isRegClass(AMDGPU::VGPR_16_LO16RegClassID);
2088bool AMDGPUOperand::isSDWAOperand(MVT type)
const {
2089 if (AsmParser->isVI())
2091 if (AsmParser->isGFX9Plus())
2092 return isRegClass(AMDGPU::VS_32RegClassID) || isInlinableImm(type);
2096bool AMDGPUOperand::isSDWAFP16Operand()
const {
2097 return isSDWAOperand(MVT::f16);
2100bool AMDGPUOperand::isSDWAFP32Operand()
const {
2101 return isSDWAOperand(MVT::f32);
2104bool AMDGPUOperand::isSDWAInt16Operand()
const {
2105 return isSDWAOperand(MVT::i16);
2108bool AMDGPUOperand::isSDWAInt32Operand()
const {
2109 return isSDWAOperand(MVT::i32);
2112bool AMDGPUOperand::isBoolReg()
const {
2113 return isReg() && ((AsmParser->isWave64() && isSCSrc_b64()) ||
2114 (AsmParser->isWave32() && isSCSrc_b32()));
2118 unsigned Size)
const {
2119 assert(isImmTy(ImmTyNone) &&
Imm.Mods.hasFPModifiers());
2134void AMDGPUOperand::addImmOperands(MCInst &Inst,
unsigned N,
2135 bool ApplyModifiers)
const {
2145 addLiteralImmOperand(Inst,
Imm.Val,
2146 ApplyModifiers & isImmTy(ImmTyNone) &&
2147 Imm.Mods.hasFPModifiers());
2149 assert(!isImmTy(ImmTyNone) || !hasModifiers());
2154void AMDGPUOperand::addLiteralImmOperand(MCInst &Inst, int64_t Val,
2155 bool ApplyModifiers)
const {
2156 const auto &InstDesc = AsmParser->getMII()->get(Inst.
getOpcode());
2161 if (ApplyModifiers) {
2163 const unsigned Size =
2165 Val = applyInputFPModifiers(Val,
Size);
2169 uint8_t OpTy = InstDesc.operands()[OpNum].OperandType;
2171 bool CanUse64BitLiterals =
2174 MCContext &Ctx = AsmParser->getContext();
2185 if (
Lit == LitModifier::None &&
2187 AsmParser->hasInv2PiInlineImm())) {
2195 bool HasMandatoryLiteral =
2198 if (
Literal.getLoBits(32) != 0 &&
2199 (InstDesc.getSize() != 4 || !AsmParser->has64BitLiterals()) &&
2200 !HasMandatoryLiteral) {
2201 const_cast<AMDGPUAsmParser *
>(AsmParser)->
Warning(
2203 "Can't encode literal as exact 64-bit floating-point operand. "
2204 "Low 32-bits will be set to zero");
2205 Val &= 0xffffffff00000000u;
2211 if (CanUse64BitLiterals &&
Lit == LitModifier::None &&
2217 Lit = LitModifier::Lit64;
2218 }
else if (
Lit == LitModifier::Lit) {
2232 if (CanUse64BitLiterals &&
Lit == LitModifier::None &&
2234 Lit = LitModifier::Lit64;
2241 if (
Lit == LitModifier::None && AsmParser->hasInv2PiInlineImm() &&
2242 Literal == 0x3fc45f306725feed) {
2282 Val = FPLiteral.bitcastToAPInt().getZExtValue();
2289 if (
Lit != LitModifier::None) {
2320 if (
Lit == LitModifier::None &&
2330 if (!AsmParser->has64BitLiterals() ||
Lit == LitModifier::Lit)
2338 if (
Lit == LitModifier::None &&
2346 if (!AsmParser->has64BitLiterals()) {
2347 Val =
static_cast<uint64_t>(Val) << 32;
2354 if (
Lit == LitModifier::Lit ||
2356 Val =
static_cast<uint64_t>(Val) << 32;
2360 if (
Lit == LitModifier::Lit)
2387 if (
Lit != LitModifier::None) {
2395void AMDGPUOperand::addRegOperands(MCInst &Inst,
unsigned N)
const {
2401bool AMDGPUOperand::isInlineValue()
const {
2409void AMDGPUAsmParser::createConstantSymbol(StringRef Id, int64_t Val) {
2420 if (Is == IS_VGPR) {
2425 return AMDGPU::VGPR_32RegClassID;
2427 return AMDGPU::VReg_64RegClassID;
2429 return AMDGPU::VReg_96RegClassID;
2431 return AMDGPU::VReg_128RegClassID;
2433 return AMDGPU::VReg_160RegClassID;
2435 return AMDGPU::VReg_192RegClassID;
2437 return AMDGPU::VReg_224RegClassID;
2439 return AMDGPU::VReg_256RegClassID;
2441 return AMDGPU::VReg_288RegClassID;
2443 return AMDGPU::VReg_320RegClassID;
2445 return AMDGPU::VReg_352RegClassID;
2447 return AMDGPU::VReg_384RegClassID;
2449 return AMDGPU::VReg_512RegClassID;
2451 return AMDGPU::VReg_1024RegClassID;
2453 }
else if (Is == IS_TTMP) {
2458 return AMDGPU::TTMP_32RegClassID;
2460 return AMDGPU::TTMP_64RegClassID;
2462 return AMDGPU::TTMP_128RegClassID;
2464 return AMDGPU::TTMP_256RegClassID;
2466 return AMDGPU::TTMP_512RegClassID;
2468 }
else if (Is == IS_SGPR) {
2473 return AMDGPU::SGPR_32RegClassID;
2475 return AMDGPU::SGPR_64RegClassID;
2477 return AMDGPU::SGPR_96RegClassID;
2479 return AMDGPU::SGPR_128RegClassID;
2481 return AMDGPU::SGPR_160RegClassID;
2483 return AMDGPU::SGPR_192RegClassID;
2485 return AMDGPU::SGPR_224RegClassID;
2487 return AMDGPU::SGPR_256RegClassID;
2489 return AMDGPU::SGPR_288RegClassID;
2491 return AMDGPU::SGPR_320RegClassID;
2493 return AMDGPU::SGPR_352RegClassID;
2495 return AMDGPU::SGPR_384RegClassID;
2497 return AMDGPU::SGPR_512RegClassID;
2499 }
else if (Is == IS_AGPR) {
2504 return AMDGPU::AGPR_32RegClassID;
2506 return AMDGPU::AReg_64RegClassID;
2508 return AMDGPU::AReg_96RegClassID;
2510 return AMDGPU::AReg_128RegClassID;
2512 return AMDGPU::AReg_160RegClassID;
2514 return AMDGPU::AReg_192RegClassID;
2516 return AMDGPU::AReg_224RegClassID;
2518 return AMDGPU::AReg_256RegClassID;
2520 return AMDGPU::AReg_288RegClassID;
2522 return AMDGPU::AReg_320RegClassID;
2524 return AMDGPU::AReg_352RegClassID;
2526 return AMDGPU::AReg_384RegClassID;
2528 return AMDGPU::AReg_512RegClassID;
2530 return AMDGPU::AReg_1024RegClassID;
2538 .
Case(
"exec", AMDGPU::EXEC)
2539 .
Case(
"vcc", AMDGPU::VCC)
2540 .
Case(
"flat_scratch", AMDGPU::FLAT_SCR)
2541 .
Case(
"xnack_mask", AMDGPU::XNACK_MASK)
2542 .
Case(
"shared_base", AMDGPU::SRC_SHARED_BASE)
2543 .
Case(
"src_shared_base", AMDGPU::SRC_SHARED_BASE)
2544 .
Case(
"shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2545 .
Case(
"src_shared_limit", AMDGPU::SRC_SHARED_LIMIT)
2546 .
Case(
"private_base", AMDGPU::SRC_PRIVATE_BASE)
2547 .
Case(
"src_private_base", AMDGPU::SRC_PRIVATE_BASE)
2548 .
Case(
"private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2549 .
Case(
"src_private_limit", AMDGPU::SRC_PRIVATE_LIMIT)
2550 .
Case(
"src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
2551 .
Case(
"src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
2552 .
Case(
"pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2553 .
Case(
"src_pops_exiting_wave_id", AMDGPU::SRC_POPS_EXITING_WAVE_ID)
2554 .
Case(
"lds_direct", AMDGPU::LDS_DIRECT)
2555 .
Case(
"src_lds_direct", AMDGPU::LDS_DIRECT)
2556 .
Case(
"m0", AMDGPU::M0)
2557 .
Case(
"vccz", AMDGPU::SRC_VCCZ)
2558 .
Case(
"src_vccz", AMDGPU::SRC_VCCZ)
2559 .
Case(
"execz", AMDGPU::SRC_EXECZ)
2560 .
Case(
"src_execz", AMDGPU::SRC_EXECZ)
2561 .
Case(
"scc", AMDGPU::SRC_SCC)
2562 .
Case(
"src_scc", AMDGPU::SRC_SCC)
2563 .
Case(
"tba", AMDGPU::TBA)
2564 .
Case(
"tma", AMDGPU::TMA)
2565 .
Case(
"flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
2566 .
Case(
"flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
2567 .
Case(
"xnack_mask_lo", AMDGPU::XNACK_MASK_LO)
2568 .
Case(
"xnack_mask_hi", AMDGPU::XNACK_MASK_HI)
2569 .
Case(
"vcc_lo", AMDGPU::VCC_LO)
2570 .
Case(
"vcc_hi", AMDGPU::VCC_HI)
2571 .
Case(
"exec_lo", AMDGPU::EXEC_LO)
2572 .
Case(
"exec_hi", AMDGPU::EXEC_HI)
2573 .
Case(
"tma_lo", AMDGPU::TMA_LO)
2574 .
Case(
"tma_hi", AMDGPU::TMA_HI)
2575 .
Case(
"tba_lo", AMDGPU::TBA_LO)
2576 .
Case(
"tba_hi", AMDGPU::TBA_HI)
2577 .
Case(
"pc", AMDGPU::PC_REG)
2578 .
Case(
"null", AMDGPU::SGPR_NULL)
2582bool AMDGPUAsmParser::ParseRegister(MCRegister &RegNo, SMLoc &StartLoc,
2583 SMLoc &EndLoc,
bool RestoreOnFailure) {
2584 auto R = parseRegister();
2588 RegNo =
R->getReg();
2589 StartLoc =
R->getStartLoc();
2590 EndLoc =
R->getEndLoc();
2594bool AMDGPUAsmParser::parseRegister(MCRegister &
Reg, SMLoc &StartLoc,
2596 return ParseRegister(
Reg, StartLoc, EndLoc,
false);
2599ParseStatus AMDGPUAsmParser::tryParseRegister(MCRegister &
Reg, SMLoc &StartLoc,
2601 bool Result = ParseRegister(
Reg, StartLoc, EndLoc,
true);
2602 bool PendingErrors = getParser().hasPendingError();
2603 getParser().clearPendingErrors();
2611bool AMDGPUAsmParser::AddNextRegisterToList(MCRegister &
Reg,
unsigned &RegWidth,
2612 RegisterKind RegKind,
2614 RegisterKind RegKind1, SMLoc Loc) {
2616 if (RegKind == IS_SGPR) {
2617 unsigned RegIdx = (
Reg - AMDGPU::SGPR0) + RegWidth / 32;
2618 if ((RegIdx == 106 && Reg1 == AMDGPU::VCC_LO) ||
2619 (RegIdx == 107 && Reg1 == AMDGPU::VCC_HI)) {
2625 if (RegKind != RegKind1) {
2626 Error(Loc,
"registers in a list must be of the same kind");
2632 if (
Reg == AMDGPU::EXEC_LO && Reg1 == AMDGPU::EXEC_HI) {
2637 if (
Reg == AMDGPU::FLAT_SCR_LO && Reg1 == AMDGPU::FLAT_SCR_HI) {
2638 Reg = AMDGPU::FLAT_SCR;
2642 if (
Reg == AMDGPU::XNACK_MASK_LO && Reg1 == AMDGPU::XNACK_MASK_HI) {
2643 Reg = AMDGPU::XNACK_MASK;
2647 if (
Reg == AMDGPU::VCC_LO && Reg1 == AMDGPU::VCC_HI) {
2652 if (
Reg == AMDGPU::TBA_LO && Reg1 == AMDGPU::TBA_HI) {
2657 if (
Reg == AMDGPU::TMA_LO && Reg1 == AMDGPU::TMA_HI) {
2662 Error(Loc,
"register does not fit in the list");
2668 if (Reg1 !=
Reg + RegWidth / 32) {
2669 Error(Loc,
"registers in a list must have consecutive indices");
2685 {{
"v"}, IS_VGPR}, {{
"s"}, IS_SGPR}, {{
"ttmp"}, IS_TTMP},
2686 {{
"acc"}, IS_AGPR}, {{
"a"}, IS_AGPR},
2690 return Kind == IS_VGPR || Kind == IS_SGPR || Kind == IS_TTMP ||
2696 if (Str.starts_with(
Reg.Name))
2702 return !Str.getAsInteger(10, Num);
2705bool AMDGPUAsmParser::isRegister(
const AsmToken &Token,
2706 const AsmToken &NextToken)
const {
2721 StringRef RegSuffix = Str.substr(
RegName.size());
2722 if (!RegSuffix.
empty()) {
2739bool AMDGPUAsmParser::isRegister() {
2740 return isRegister(
getToken(), peekToken());
2743MCRegister AMDGPUAsmParser::getRegularReg(RegisterKind RegKind,
unsigned RegNum,
2744 unsigned SubReg,
unsigned RegWidth,
2748 unsigned AlignSize = 1;
2749 if (RegKind == IS_SGPR || RegKind == IS_TTMP) {
2755 if (RegNum % AlignSize != 0) {
2756 Error(Loc,
"invalid register alignment");
2757 return MCRegister();
2760 unsigned RegIdx = RegNum / AlignSize;
2763 Error(Loc,
"invalid or unsupported register size");
2764 return MCRegister();
2768 const MCRegisterClass &RC =
TRI->getRegClass(RCID);
2769 if (RegIdx >= RC.
getNumRegs() || (RegKind == IS_VGPR && RegIdx > 255)) {
2770 Error(Loc,
"register index is out of range");
2771 return AMDGPU::NoRegister;
2774 if (RegKind == IS_VGPR && !
isGFX1250Plus() && RegIdx + RegWidth / 32 > 256) {
2775 Error(Loc,
"register index is out of range");
2776 return MCRegister();
2785 Error(Loc,
"invalid subregister");
2791bool AMDGPUAsmParser::ParseRegRange(
unsigned &Num,
unsigned &RegWidth,
2793 int64_t RegLo, RegHi;
2797 SMLoc FirstIdxLoc = getLoc();
2804 SecondIdxLoc = getLoc();
2815 Error(FirstIdxLoc,
"invalid register index");
2820 Error(SecondIdxLoc,
"invalid register index");
2824 if (RegLo > RegHi) {
2825 Error(FirstIdxLoc,
"first register index should not exceed second index");
2829 if (RegHi == RegLo) {
2830 StringRef RegSuffix = getTokenStr();
2831 if (RegSuffix ==
".l") {
2832 SubReg = AMDGPU::lo16;
2834 }
else if (RegSuffix ==
".h") {
2835 SubReg = AMDGPU::hi16;
2840 Num =
static_cast<unsigned>(RegLo);
2841 RegWidth = 32 * ((RegHi - RegLo) + 1);
2846MCRegister AMDGPUAsmParser::ParseSpecialReg(RegisterKind &RegKind,
2849 SmallVectorImpl<AsmToken> &Tokens) {
2855 RegKind = IS_SPECIAL;
2862MCRegister AMDGPUAsmParser::ParseRegularReg(RegisterKind &RegKind,
2865 SmallVectorImpl<AsmToken> &Tokens) {
2867 StringRef
RegName = getTokenStr();
2868 auto Loc = getLoc();
2872 Error(Loc,
"invalid register name");
2873 return MCRegister();
2881 unsigned SubReg = NoSubRegister;
2882 bool IsRange =
false;
2883 if (!RegSuffix.
empty()) {
2885 SubReg = AMDGPU::lo16;
2887 SubReg = AMDGPU::hi16;
2891 Error(Loc,
"invalid register index");
2892 return MCRegister();
2898 if (!ParseRegRange(RegNum, RegWidth, SubReg))
2899 return MCRegister();
2903 MCRegister
Reg = getRegularReg(RegKind, RegNum, SubReg, RegWidth, Loc);
2904 const MCRegisterInfo &
TRI = *
getContext().getRegisterInfo();
2905 if (RegKind == IS_SGPR && IsRange
2906 ? (
TRI.isSubRegister(
Reg, VCC_LO) ||
TRI.isSubRegister(
Reg, VCC_HI))
2907 : (
Reg == VCC_LO ||
Reg == VCC_HI)) {
2908 Error(Loc,
"register index is out of range");
2909 return MCRegister();
2915MCRegister AMDGPUAsmParser::ParseRegList(RegisterKind &RegKind,
2916 unsigned &RegNum,
unsigned &RegWidth,
2917 SmallVectorImpl<AsmToken> &Tokens) {
2919 auto ListLoc = getLoc();
2922 "expected a register or a list of registers")) {
2923 return MCRegister();
2928 auto Loc = getLoc();
2929 if (!ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth))
2930 return MCRegister();
2931 if (RegWidth != 32) {
2932 Error(Loc,
"expected a single 32-bit register");
2933 return MCRegister();
2937 RegisterKind NextRegKind;
2939 unsigned NextRegNum, NextRegWidth;
2942 if (!ParseAMDGPURegister(NextRegKind, NextReg, NextRegNum, NextRegWidth,
2944 return MCRegister();
2946 if (NextRegWidth != 32) {
2947 Error(Loc,
"expected a single 32-bit register");
2948 return MCRegister();
2950 if (!AddNextRegisterToList(
Reg, RegWidth, RegKind, NextReg, NextRegKind,
2952 return MCRegister();
2956 "expected a comma or a closing square bracket")) {
2957 return MCRegister();
2961 Reg = getRegularReg(RegKind, RegNum, NoSubRegister, RegWidth, ListLoc);
2966bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
2967 MCRegister &
Reg,
unsigned &RegNum,
2969 SmallVectorImpl<AsmToken> &Tokens) {
2970 auto Loc = getLoc();
2974 Reg = ParseSpecialReg(RegKind, RegNum, RegWidth, Tokens);
2976 Reg = ParseRegularReg(RegKind, RegNum, RegWidth, Tokens);
2978 Reg = ParseRegList(RegKind, RegNum, RegWidth, Tokens);
2983 assert(Parser.hasPendingError());
2987 if (!subtargetHasRegister(*
TRI,
Reg)) {
2988 if (
Reg == AMDGPU::SGPR_NULL) {
2989 Error(Loc,
"'null' operand is not supported on this GPU");
2992 " register not available on this GPU");
3000bool AMDGPUAsmParser::ParseAMDGPURegister(RegisterKind &RegKind,
3001 MCRegister &
Reg,
unsigned &RegNum,
3003 bool RestoreOnFailure ) {
3007 if (ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth, Tokens)) {
3008 if (RestoreOnFailure) {
3009 while (!Tokens.
empty()) {
3018std::optional<StringRef>
3019AMDGPUAsmParser::getGprCountSymbolName(RegisterKind RegKind) {
3022 return StringRef(
".amdgcn.next_free_vgpr");
3024 return StringRef(
".amdgcn.next_free_sgpr");
3026 return std::nullopt;
3030void AMDGPUAsmParser::initializeGprCountSymbol(RegisterKind RegKind) {
3031 auto SymbolName = getGprCountSymbolName(RegKind);
3032 assert(SymbolName &&
"initializing invalid register kind");
3038bool AMDGPUAsmParser::updateGprCountSymbols(RegisterKind RegKind,
3039 unsigned DwordRegIndex,
3040 unsigned RegWidth) {
3045 auto SymbolName = getGprCountSymbolName(RegKind);
3050 int64_t NewMax = DwordRegIndex +
divideCeil(RegWidth, 32) - 1;
3054 return !
Error(getLoc(),
3055 ".amdgcn.next_free_{v,s}gpr symbols must be variable");
3059 ".amdgcn.next_free_{v,s}gpr symbols must be absolute expressions");
3061 if (OldCount <= NewMax)
3067std::unique_ptr<AMDGPUOperand>
3068AMDGPUAsmParser::parseRegister(
bool RestoreOnFailure) {
3070 SMLoc StartLoc = Tok.getLoc();
3071 SMLoc EndLoc = Tok.getEndLoc();
3072 RegisterKind RegKind;
3074 unsigned RegNum, RegWidth;
3076 if (!ParseAMDGPURegister(RegKind,
Reg, RegNum, RegWidth)) {
3080 if (!updateGprCountSymbols(RegKind, RegNum, RegWidth))
3083 KernelScope.usesRegister(RegKind, RegNum, RegWidth);
3084 return AMDGPUOperand::CreateReg(
this,
Reg, StartLoc, EndLoc);
3091 if (isRegister() || isModifier())
3094 if (
Lit == LitModifier::None) {
3095 if (trySkipId(
"lit"))
3096 Lit = LitModifier::Lit;
3097 else if (trySkipId(
"lit64"))
3098 Lit = LitModifier::Lit64;
3100 if (
Lit != LitModifier::None) {
3103 ParseStatus S = parseImm(
Operands, HasSP3AbsModifier,
Lit);
3112 const auto &NextTok = peekToken();
3115 bool Negate =
false;
3123 AMDGPUOperand::Modifiers Mods;
3131 StringRef Num = getTokenStr();
3134 APFloat RealVal(APFloat::IEEEdouble());
3135 auto roundMode = APFloat::rmNearestTiesToEven;
3136 if (
errorToBool(RealVal.convertFromString(Num, roundMode).takeError()))
3139 RealVal.changeSign();
3142 AMDGPUOperand::CreateImm(
this, RealVal.bitcastToAPInt().getZExtValue(),
3143 S, AMDGPUOperand::ImmTyNone,
true));
3144 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3145 Op.setModifiers(Mods);
3154 if (HasSP3AbsModifier) {
3163 if (getParser().parsePrimaryExpr(Expr, EndLoc,
nullptr))
3166 if (Parser.parseExpression(Expr))
3170 if (Expr->evaluateAsAbsolute(IntVal)) {
3172 return Error(S,
"literal value out of range");
3173 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S));
3174 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3175 Op.setModifiers(Mods);
3177 if (
Lit != LitModifier::None)
3179 Operands.push_back(AMDGPUOperand::CreateExpr(
this, Expr, S));
3192 if (
auto R = parseRegister()) {
3202 ParseStatus Res = parseReg(
Operands);
3210bool AMDGPUAsmParser::isNamedOperandModifier(
const AsmToken &Token,
3211 const AsmToken &NextToken)
const {
3214 return str ==
"abs" || str ==
"neg" || str ==
"sext";
3219bool AMDGPUAsmParser::isOpcodeModifierWithVal(
const AsmToken &Token,
3220 const AsmToken &NextToken)
const {
3224bool AMDGPUAsmParser::isOperandModifier(
const AsmToken &Token,
3225 const AsmToken &NextToken)
const {
3226 return isNamedOperandModifier(Token, NextToken) || Token.
is(
AsmToken::Pipe);
3229bool AMDGPUAsmParser::isRegOrOperandModifier(
const AsmToken &Token,
3230 const AsmToken &NextToken)
const {
3231 return isRegister(Token, NextToken) || isOperandModifier(Token, NextToken);
3248bool AMDGPUAsmParser::isModifier() {
3251 AsmToken NextToken[2];
3252 peekTokens(NextToken);
3256 bool IsOpcodeModifier = isOpcodeModifierWithVal(Tok, NextToken[0]) &&
3259 return isOperandModifier(Tok, NextToken[0]) ||
3261 isRegOrOperandModifier(NextToken[0], NextToken[1])) ||
3287bool AMDGPUAsmParser::parseSP3NegModifier() {
3289 AsmToken NextToken[2];
3290 peekTokens(NextToken);
3293 (isRegister(NextToken[0], NextToken[1]) ||
3311 return Error(getLoc(),
"invalid syntax, expected 'neg' modifier");
3313 SP3Neg = parseSP3NegModifier();
3316 Neg = trySkipId(
"neg");
3318 return Error(Loc,
"expected register or immediate");
3322 Abs = trySkipId(
"abs");
3327 if (trySkipId(
"lit")) {
3328 Lit = LitModifier::Lit;
3331 }
else if (trySkipId(
"lit64")) {
3332 Lit = LitModifier::Lit64;
3335 if (!has64BitLiterals())
3336 return Error(Loc,
"lit64 is not supported on this GPU");
3342 return Error(Loc,
"expected register or immediate");
3351 return (SP3Neg || Neg || SP3Abs || Abs ||
Lit != LitModifier::None)
3355 if (
Lit != LitModifier::None && !
Operands.back()->isImm())
3356 Error(Loc,
"expected immediate with lit modifier");
3358 if (SP3Abs && !skipToken(
AsmToken::Pipe,
"expected vertical bar"))
3364 if (
Lit != LitModifier::None &&
3368 AMDGPUOperand::Modifiers Mods;
3369 Mods.Abs = Abs || SP3Abs;
3370 Mods.Neg = Neg || SP3Neg;
3373 if (Mods.hasFPModifiers() ||
Lit != LitModifier::None) {
3374 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3376 return Error(
Op.getStartLoc(),
"expected an absolute expression");
3377 Op.setModifiers(Mods);
3385 bool Sext = trySkipId(
"sext");
3386 if (Sext && !skipToken(
AsmToken::LParen,
"expected left paren after sext"))
3401 AMDGPUOperand::Modifiers Mods;
3404 if (Mods.hasIntModifiers()) {
3405 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands.back());
3407 return Error(
Op.getStartLoc(),
"expected an absolute expression");
3408 Op.setModifiers(Mods);
3415 return parseRegOrImmWithFPInputMods(
Operands,
false);
3419 return parseRegOrImmWithIntInputMods(
Operands,
false);
3426 if (!trySkipId(
"rsrcidx"))
3432 SMLoc RegLoc = getLoc();
3433 std::unique_ptr<AMDGPUOperand>
Reg = parseRegister();
3441 if (!
Reg->isRsrcReg32())
3442 return Error(RegLoc,
"rsrcidx operand must be a 32-bit SGPR or VGPR");
3452 auto Loc = getLoc();
3453 if (trySkipId(
"off")) {
3455 AMDGPUOperand::CreateImm(
this, 0, Loc, AMDGPUOperand::ImmTyOff,
false));
3462 std::unique_ptr<AMDGPUOperand>
Reg = parseRegister();
3471unsigned AMDGPUAsmParser::checkTargetMatchPredicate(MCInst &Inst) {
3476 return Match_InvalidOperand;
3478 if (Inst.
getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
3479 Inst.
getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
3482 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::dst_sel);
3484 if (!
Op.isImm() ||
Op.getImm() != AMDGPU::SDWA::SdwaSel::DWORD) {
3485 return Match_InvalidOperand;
3493 if (tryAnotherVOPDEncoding(Inst))
3494 return Match_InvalidOperand;
3496 return Match_Success;
3500 static const unsigned Variants[] = {
3509ArrayRef<unsigned> AMDGPUAsmParser::getMatchedVariants()
const {
3510 if (isForcedDPP() && isForcedVOP3()) {
3514 if (getForcedEncodingSize() == 32) {
3519 if (isForcedVOP3()) {
3524 if (isForcedSDWA()) {
3530 if (isForcedDPP()) {
3538StringRef AMDGPUAsmParser::getMatchedVariantName()
const {
3539 if (isForcedDPP() && isForcedVOP3())
3542 if (getForcedEncodingSize() == 32)
3558AMDGPUAsmParser::findImplicitSGPRReadInVOP(
const MCInst &Inst)
const {
3562 case AMDGPU::FLAT_SCR:
3564 case AMDGPU::VCC_LO:
3565 case AMDGPU::VCC_HI:
3572 return MCRegister();
3579bool AMDGPUAsmParser::isInlineConstant(
const MCInst &Inst,
3580 unsigned OpIdx)
const {
3588 const MCOperand &MO = Inst.
getOperand(OpIdx);
3638unsigned AMDGPUAsmParser::getConstantBusLimit(
unsigned Opcode)
const {
3644 case AMDGPU::V_LSHLREV_B64_e64:
3645 case AMDGPU::V_LSHLREV_B64_gfx10:
3646 case AMDGPU::V_LSHLREV_B64_e64_gfx11:
3647 case AMDGPU::V_LSHLREV_B64_e32_gfx12:
3648 case AMDGPU::V_LSHLREV_B64_e64_gfx12:
3649 case AMDGPU::V_LSHRREV_B64_e64:
3650 case AMDGPU::V_LSHRREV_B64_gfx10:
3651 case AMDGPU::V_LSHRREV_B64_e64_gfx11:
3652 case AMDGPU::V_LSHRREV_B64_e64_gfx12:
3653 case AMDGPU::V_ASHRREV_I64_e64:
3654 case AMDGPU::V_ASHRREV_I64_gfx10:
3655 case AMDGPU::V_ASHRREV_I64_e64_gfx11:
3656 case AMDGPU::V_ASHRREV_I64_e64_gfx12:
3657 case AMDGPU::V_LSHL_B64_e64:
3658 case AMDGPU::V_LSHR_B64_e64:
3659 case AMDGPU::V_ASHR_I64_e64:
3672 bool AddMandatoryLiterals =
false) {
3675 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::imm) : -1;
3679 AddMandatoryLiterals ? getNamedOperandIdx(Opcode, OpName::immX) : -1;
3681 return {getNamedOperandIdx(Opcode, OpName::src0X),
3682 getNamedOperandIdx(Opcode, OpName::vsrc1X),
3683 getNamedOperandIdx(Opcode, OpName::vsrc2X),
3684 getNamedOperandIdx(Opcode, OpName::src0Y),
3685 getNamedOperandIdx(Opcode, OpName::vsrc1Y),
3686 getNamedOperandIdx(Opcode, OpName::vsrc2Y),
3691 return {getNamedOperandIdx(Opcode, OpName::src0),
3692 getNamedOperandIdx(Opcode, OpName::src1),
3693 getNamedOperandIdx(Opcode, OpName::src2), ImmIdx};
3696bool AMDGPUAsmParser::usesConstantBus(
const MCInst &Inst,
unsigned OpIdx) {
3697 const MCOperand &MO = Inst.
getOperand(OpIdx);
3699 return !isInlineConstant(Inst, OpIdx);
3706 return isSGPR(PReg,
TRI) && PReg != SGPR_NULL;
3717 const unsigned Opcode = Inst.
getOpcode();
3718 if (Opcode != V_WRITELANE_B32_gfx6_gfx7 && Opcode != V_WRITELANE_B32_vi)
3721 if (!LaneSelOp.
isReg())
3724 return LaneSelReg ==
M0 || LaneSelReg == M0_gfxpre11;
3727bool AMDGPUAsmParser::validateConstantBusLimitations(
3729 const unsigned Opcode = Inst.
getOpcode();
3730 const MCInstrDesc &
Desc = MII.
get(Opcode);
3731 MCRegister LastSGPR;
3732 unsigned ConstantBusUseCount = 0;
3733 unsigned NumLiterals = 0;
3734 unsigned LiteralSize;
3750 SmallDenseSet<MCRegister> SGPRsUsed;
3751 MCRegister SGPRUsed = findImplicitSGPRReadInVOP(Inst);
3753 SGPRsUsed.
insert(SGPRUsed);
3754 ++ConstantBusUseCount;
3759 unsigned ConstantBusLimit = getConstantBusLimit(Opcode);
3761 for (
int OpIdx : OpIndices) {
3765 const MCOperand &MO = Inst.
getOperand(OpIdx);
3766 if (usesConstantBus(Inst, OpIdx)) {
3775 if (SGPRsUsed.
insert(LastSGPR).second) {
3776 ++ConstantBusUseCount;
3796 if (NumLiterals == 0) {
3799 }
else if (LiteralSize !=
Size) {
3805 if (ConstantBusUseCount + NumLiterals > ConstantBusLimit) {
3807 "invalid operand (violates constant bus restrictions)");
3814std::optional<unsigned>
3815AMDGPUAsmParser::checkVOPDRegBankConstraints(
const MCInst &Inst,
bool AsVOPD3) {
3817 const unsigned Opcode = Inst.
getOpcode();
3823 auto getVRegIdx = [&](unsigned,
unsigned OperandIdx) {
3824 const MCOperand &Opr = Inst.
getOperand(OperandIdx);
3833 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1170 ||
3834 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx12 ||
3835 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx1250 ||
3836 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_gfx13 ||
3837 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx1250 ||
3838 Opcode == AMDGPU::V_DUAL_MOV_B32_e32_X_MOV_B32_e32_e96_gfx13;
3842 for (
auto OpName : {OpName::src0X, OpName::src0Y}) {
3843 int I = getNamedOperandIdx(Opcode, OpName);
3847 int64_t
Imm =
Op.getImm();
3853 for (
auto OpName : {OpName::vsrc1X, OpName::vsrc1Y, OpName::vsrc2X,
3854 OpName::vsrc2Y, OpName::imm}) {
3855 int I = getNamedOperandIdx(Opcode, OpName);
3865 auto InvalidCompOprIdx = InstInfo.getInvalidCompOperandIndex(
3866 getVRegIdx, *
TRI, SkipSrc, AllowSameVGPR, AsVOPD3);
3868 return InvalidCompOprIdx;
3871bool AMDGPUAsmParser::validateVOPD(
const MCInst &Inst,
3878 for (
const std::unique_ptr<MCParsedAsmOperand> &Operand :
Operands) {
3879 AMDGPUOperand &
Op = (AMDGPUOperand &)*Operand;
3880 if ((
Op.isRegKind() ||
Op.isImmTy(AMDGPUOperand::ImmTyNone)) &&
3882 Error(
Op.getStartLoc(),
"ABS not allowed in VOPD3 instructions");
3886 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst, AsVOPD3);
3887 if (!InvalidCompOprIdx.has_value())
3890 auto CompOprIdx = *InvalidCompOprIdx;
3893 std::max(InstInfo[
VOPD::X].getIndexInParsedOperands(CompOprIdx),
3894 InstInfo[
VOPD::Y].getIndexInParsedOperands(CompOprIdx));
3897 auto Loc = ((AMDGPUOperand &)*
Operands[ParsedIdx]).getStartLoc();
3898 if (CompOprIdx == VOPD::Component::DST) {
3900 Error(Loc,
"dst registers must be distinct");
3902 Error(Loc,
"one dst register must be even and the other odd");
3904 auto CompSrcIdx = CompOprIdx - VOPD::Component::DST_NUM;
3905 Error(Loc, Twine(
"src") + Twine(CompSrcIdx) +
3906 " operands must use different VGPR banks");
3914bool AMDGPUAsmParser::tryVOPD3(
const MCInst &Inst) {
3916 auto InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst,
false);
3917 if (!InvalidCompOprIdx.has_value())
3921 InvalidCompOprIdx = checkVOPDRegBankConstraints(Inst,
true);
3922 if (InvalidCompOprIdx.has_value()) {
3927 if (*InvalidCompOprIdx == VOPD::Component::DST)
3940bool AMDGPUAsmParser::tryVOPD(
const MCInst &Inst) {
3941 const unsigned Opcode = Inst.
getOpcode();
3956 for (
auto OpName : {OpName::src0X_modifiers, OpName::src0Y_modifiers,
3957 OpName::vsrc1X_modifiers, OpName::vsrc1Y_modifiers,
3958 OpName::vsrc2X_modifiers, OpName::vsrc2Y_modifiers}) {
3959 int I = getNamedOperandIdx(Opcode, OpName);
3966 return !tryVOPD3(Inst);
3971bool AMDGPUAsmParser::tryAnotherVOPDEncoding(
const MCInst &Inst) {
3976 return tryVOPD(Inst);
3977 return tryVOPD3(Inst);
3980bool AMDGPUAsmParser::validateIntClampSupported(
const MCInst &Inst) {
3985 int ClampIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::clamp);
3993bool AMDGPUAsmParser::validateMIMGDataSize(
const MCInst &Inst, SMLoc IDLoc) {
4001 int VDataIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdata);
4002 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4003 int TFEIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::tfe);
4008 if ((DMaskIdx == -1 || TFEIdx == -1) &&
4009 hasBVHRayTracingInsts())
4012 unsigned VDataSize = getRegOperandSize(
Desc, VDataIdx);
4013 unsigned TFESize = (TFEIdx != -1 && Inst.
getOperand(TFEIdx).
getImm()) ? 1 : 0;
4018 bool IsPackedD16 =
false;
4021 int D16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::d16);
4022 IsPackedD16 = D16Idx >= 0;
4024 DataSize = (DataSize + 1) / 2;
4027 if ((VDataSize / 4) == DataSize + TFESize)
4032 Modifiers = IsPackedD16 ?
"dmask and d16" :
"dmask";
4034 Modifiers = IsPackedD16 ?
"dmask, d16 and tfe" :
"dmask and tfe";
4036 Error(IDLoc, Twine(
"image data size does not match ") + Modifiers);
4040bool AMDGPUAsmParser::validateMIMGAddrSize(
const MCInst &Inst, SMLoc IDLoc) {
4049 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4051 int VAddr0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr0);
4052 AMDGPU::OpName RSrcOpName =
4054 int SrsrcIdx = AMDGPU::getNamedOperandIdx(
Opc, RSrcOpName);
4055 int DimIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dim);
4056 int A16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::a16);
4060 assert(SrsrcIdx > VAddr0Idx);
4063 if (BaseOpcode->
BVH) {
4064 if (IsA16 == BaseOpcode->
A16)
4066 Error(IDLoc,
"image address size does not match a16");
4072 bool IsNSA = SrsrcIdx - VAddr0Idx > 1;
4073 unsigned ActualAddrSize =
4074 IsNSA ? SrsrcIdx - VAddr0Idx : getRegOperandSize(
Desc, VAddr0Idx) / 4;
4076 unsigned ExpectedAddrSize =
4080 if (hasPartialNSAEncoding() &&
4082 int VAddrLastIdx = SrsrcIdx - 1;
4083 unsigned VAddrLastSize = getRegOperandSize(
Desc, VAddrLastIdx) / 4;
4085 ActualAddrSize = VAddrLastIdx - VAddr0Idx + VAddrLastSize;
4088 if (ExpectedAddrSize > 12)
4089 ExpectedAddrSize = 16;
4094 if (ActualAddrSize == 8 && (ExpectedAddrSize >= 5 && ExpectedAddrSize <= 7))
4098 if (ActualAddrSize == ExpectedAddrSize)
4101 Error(IDLoc,
"image address size does not match dim and a16");
4105bool AMDGPUAsmParser::validateMIMGAtomicDMask(
const MCInst &Inst) {
4112 if (!
Desc.mayLoad() || !
Desc.mayStore())
4115 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4122 return DMask == 0x1 || DMask == 0x3 || DMask == 0xf;
4125bool AMDGPUAsmParser::validateMIMGGatherDMask(
const MCInst &Inst) {
4132 int DMaskIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dmask);
4140 return DMask == 0x1 || DMask == 0x2 || DMask == 0x4 || DMask == 0x8;
4143bool AMDGPUAsmParser::validateMIMGDim(
const MCInst &Inst,
4157 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
4158 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4165bool AMDGPUAsmParser::validateMIMGMSAA(
const MCInst &Inst) {
4172 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
4175 if (!BaseOpcode->
MSAA)
4178 int DimIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dim);
4184 return DimInfo->
MSAA;
4189 case AMDGPU::V_MOVRELS_B32_sdwa_gfx10:
4190 case AMDGPU::V_MOVRELSD_B32_sdwa_gfx10:
4191 case AMDGPU::V_MOVRELSD_2_B32_sdwa_gfx10:
4201bool AMDGPUAsmParser::validateMovrels(
const MCInst &Inst,
4209 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4220 Error(getOperandLoc(
Operands, Src0Idx),
"source operand must be a VGPR");
4224bool AMDGPUAsmParser::validateMAIAccWrite(
const MCInst &Inst,
4229 if (
Opc != AMDGPU::V_ACCVGPR_WRITE_B32_vi)
4232 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4243 "source operand must be either a VGPR or an inline constant");
4250bool AMDGPUAsmParser::validateMAISrc2(
const MCInst &Inst,
4255 !getFeatureBits()[FeatureMFMAInlineLiteralBug])
4258 const int Src2Idx = getNamedOperandIdx(Opcode, OpName::src2);
4262 if (Inst.
getOperand(Src2Idx).
isImm() && isInlineConstant(Inst, Src2Idx)) {
4264 "inline constants are not allowed for this operand");
4271bool AMDGPUAsmParser::validateMFMA(
const MCInst &Inst,
4279 int BlgpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
4280 if (BlgpIdx != -1) {
4281 if (
const MFMA_F8F6F4_Info *Info = AMDGPU::isMFMA_F8F6F4(
Opc)) {
4282 int CbszIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::cbsz);
4292 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4294 "wrong register tuple size for cbsz value " + Twine(CBSZ));
4299 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
4301 "wrong register tuple size for blgp value " + Twine(BLGP));
4309 const int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
4319 if (Src2Reg == DstReg)
4324 .getSizeInBits() <= 128)
4327 if (
TRI->regsOverlap(Src2Reg, DstReg)) {
4329 "source 2 operand must not partially overlap with dst");
4336bool AMDGPUAsmParser::validateDivScale(
const MCInst &Inst) {
4340 case V_DIV_SCALE_F32_gfx6_gfx7:
4341 case V_DIV_SCALE_F32_vi:
4342 case V_DIV_SCALE_F32_gfx10:
4343 case V_DIV_SCALE_F64_gfx6_gfx7:
4344 case V_DIV_SCALE_F64_vi:
4345 case V_DIV_SCALE_F64_gfx10:
4352 {AMDGPU::OpName::src0_modifiers, AMDGPU::OpName::src2_modifiers,
4353 AMDGPU::OpName::src2_modifiers}) {
4364bool AMDGPUAsmParser::validateMIMGD16(
const MCInst &Inst) {
4371 int D16Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::d16);
4380bool AMDGPUAsmParser::validateTensorR128(
const MCInst &Inst) {
4386 int R128Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::r128);
4393 case AMDGPU::V_SUBREV_F32_e32:
4394 case AMDGPU::V_SUBREV_F32_e64:
4395 case AMDGPU::V_SUBREV_F32_e32_gfx10:
4396 case AMDGPU::V_SUBREV_F32_e32_gfx6_gfx7:
4397 case AMDGPU::V_SUBREV_F32_e32_vi:
4398 case AMDGPU::V_SUBREV_F32_e64_gfx10:
4399 case AMDGPU::V_SUBREV_F32_e64_gfx6_gfx7:
4400 case AMDGPU::V_SUBREV_F32_e64_vi:
4402 case AMDGPU::V_SUBREV_CO_U32_e32:
4403 case AMDGPU::V_SUBREV_CO_U32_e64:
4404 case AMDGPU::V_SUBREV_I32_e32_gfx6_gfx7:
4405 case AMDGPU::V_SUBREV_I32_e64_gfx6_gfx7:
4407 case AMDGPU::V_SUBBREV_U32_e32:
4408 case AMDGPU::V_SUBBREV_U32_e64:
4409 case AMDGPU::V_SUBBREV_U32_e32_gfx6_gfx7:
4410 case AMDGPU::V_SUBBREV_U32_e32_vi:
4411 case AMDGPU::V_SUBBREV_U32_e64_gfx6_gfx7:
4412 case AMDGPU::V_SUBBREV_U32_e64_vi:
4414 case AMDGPU::V_SUBREV_U32_e32:
4415 case AMDGPU::V_SUBREV_U32_e64:
4416 case AMDGPU::V_SUBREV_U32_e32_gfx9:
4417 case AMDGPU::V_SUBREV_U32_e32_vi:
4418 case AMDGPU::V_SUBREV_U32_e64_gfx9:
4419 case AMDGPU::V_SUBREV_U32_e64_vi:
4421 case AMDGPU::V_SUBREV_F16_e32:
4422 case AMDGPU::V_SUBREV_F16_e64:
4423 case AMDGPU::V_SUBREV_F16_e32_gfx10:
4424 case AMDGPU::V_SUBREV_F16_e32_vi:
4425 case AMDGPU::V_SUBREV_F16_e64_gfx10:
4426 case AMDGPU::V_SUBREV_F16_e64_vi:
4428 case AMDGPU::V_SUBREV_U16_e32:
4429 case AMDGPU::V_SUBREV_U16_e64:
4430 case AMDGPU::V_SUBREV_U16_e32_vi:
4431 case AMDGPU::V_SUBREV_U16_e64_vi:
4433 case AMDGPU::V_SUBREV_CO_U32_e32_gfx9:
4434 case AMDGPU::V_SUBREV_CO_U32_e64_gfx10:
4435 case AMDGPU::V_SUBREV_CO_U32_e64_gfx9:
4437 case AMDGPU::V_SUBBREV_CO_U32_e32_gfx9:
4438 case AMDGPU::V_SUBBREV_CO_U32_e64_gfx9:
4440 case AMDGPU::V_SUBREV_NC_U32_e32_gfx10:
4441 case AMDGPU::V_SUBREV_NC_U32_e64_gfx10:
4443 case AMDGPU::V_SUBREV_CO_CI_U32_e32_gfx10:
4444 case AMDGPU::V_SUBREV_CO_CI_U32_e64_gfx10:
4446 case AMDGPU::V_LSHRREV_B32_e32:
4447 case AMDGPU::V_LSHRREV_B32_e64:
4448 case AMDGPU::V_LSHRREV_B32_e32_gfx6_gfx7:
4449 case AMDGPU::V_LSHRREV_B32_e64_gfx6_gfx7:
4450 case AMDGPU::V_LSHRREV_B32_e32_vi:
4451 case AMDGPU::V_LSHRREV_B32_e64_vi:
4452 case AMDGPU::V_LSHRREV_B32_e32_gfx10:
4453 case AMDGPU::V_LSHRREV_B32_e64_gfx10:
4455 case AMDGPU::V_ASHRREV_I32_e32:
4456 case AMDGPU::V_ASHRREV_I32_e64:
4457 case AMDGPU::V_ASHRREV_I32_e32_gfx10:
4458 case AMDGPU::V_ASHRREV_I32_e32_gfx6_gfx7:
4459 case AMDGPU::V_ASHRREV_I32_e32_vi:
4460 case AMDGPU::V_ASHRREV_I32_e64_gfx10:
4461 case AMDGPU::V_ASHRREV_I32_e64_gfx6_gfx7:
4462 case AMDGPU::V_ASHRREV_I32_e64_vi:
4464 case AMDGPU::V_LSHLREV_B32_e32:
4465 case AMDGPU::V_LSHLREV_B32_e64:
4466 case AMDGPU::V_LSHLREV_B32_e32_gfx10:
4467 case AMDGPU::V_LSHLREV_B32_e32_gfx6_gfx7:
4468 case AMDGPU::V_LSHLREV_B32_e32_vi:
4469 case AMDGPU::V_LSHLREV_B32_e64_gfx10:
4470 case AMDGPU::V_LSHLREV_B32_e64_gfx6_gfx7:
4471 case AMDGPU::V_LSHLREV_B32_e64_vi:
4473 case AMDGPU::V_LSHLREV_B16_e32:
4474 case AMDGPU::V_LSHLREV_B16_e64:
4475 case AMDGPU::V_LSHLREV_B16_e32_vi:
4476 case AMDGPU::V_LSHLREV_B16_e64_vi:
4477 case AMDGPU::V_LSHLREV_B16_gfx10:
4479 case AMDGPU::V_LSHRREV_B16_e32:
4480 case AMDGPU::V_LSHRREV_B16_e64:
4481 case AMDGPU::V_LSHRREV_B16_e32_vi:
4482 case AMDGPU::V_LSHRREV_B16_e64_vi:
4483 case AMDGPU::V_LSHRREV_B16_gfx10:
4485 case AMDGPU::V_ASHRREV_I16_e32:
4486 case AMDGPU::V_ASHRREV_I16_e64:
4487 case AMDGPU::V_ASHRREV_I16_e32_vi:
4488 case AMDGPU::V_ASHRREV_I16_e64_vi:
4489 case AMDGPU::V_ASHRREV_I16_gfx10:
4491 case AMDGPU::V_LSHLREV_B64_e64:
4492 case AMDGPU::V_LSHLREV_B64_gfx10:
4493 case AMDGPU::V_LSHLREV_B64_vi:
4495 case AMDGPU::V_LSHRREV_B64_e64:
4496 case AMDGPU::V_LSHRREV_B64_gfx10:
4497 case AMDGPU::V_LSHRREV_B64_vi:
4499 case AMDGPU::V_ASHRREV_I64_e64:
4500 case AMDGPU::V_ASHRREV_I64_gfx10:
4501 case AMDGPU::V_ASHRREV_I64_vi:
4503 case AMDGPU::V_PK_LSHLREV_B16:
4504 case AMDGPU::V_PK_LSHLREV_B16_gfx10:
4505 case AMDGPU::V_PK_LSHLREV_B16_vi:
4507 case AMDGPU::V_PK_LSHRREV_B16:
4508 case AMDGPU::V_PK_LSHRREV_B16_gfx10:
4509 case AMDGPU::V_PK_LSHRREV_B16_vi:
4510 case AMDGPU::V_PK_ASHRREV_I16:
4511 case AMDGPU::V_PK_ASHRREV_I16_gfx10:
4512 case AMDGPU::V_PK_ASHRREV_I16_vi:
4519bool AMDGPUAsmParser::validateLdsDirect(
const MCInst &Inst,
4521 const unsigned Opcode = Inst.
getOpcode();
4530 for (
auto SrcName : {OpName::src0, OpName::src1, OpName::src2}) {
4531 auto SrcIdx = getNamedOperandIdx(Opcode, SrcName);
4535 if (Src.isReg() && Src.getReg() == LDS_DIRECT) {
4539 "lds_direct is not supported on this GPU");
4545 "lds_direct cannot be used with this instruction");
4549 if (SrcName != OpName::src0) {
4551 "lds_direct may be used as src0 only");
4561 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
4562 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4563 if (
Op.isFlatOffset())
4564 return Op.getStartLoc();
4569bool AMDGPUAsmParser::validateOffset(
const MCInst &Inst,
4572 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4577 return validateFlatOffset(Inst,
Operands);
4580 return validateSMEMOffset(Inst,
Operands);
4585 const unsigned OffsetSize = 24;
4586 if (!
isUIntN(OffsetSize - 1,
Op.getImm())) {
4588 Twine(
"expected a ") + Twine(OffsetSize - 1) +
4589 "-bit unsigned offset for buffer ops");
4593 const unsigned OffsetSize = 16;
4594 if (!
isUIntN(OffsetSize,
Op.getImm())) {
4596 Twine(
"expected a ") + Twine(OffsetSize) +
"-bit unsigned offset");
4603bool AMDGPUAsmParser::validateFlatOffset(
const MCInst &Inst,
4609 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4613 if (!hasFlatOffsets() &&
Op.getImm() != 0) {
4615 "flat offset modifier is not supported on this GPU");
4622 bool AllowNegative =
4624 if (!
isIntN(OffsetSize,
Op.getImm()) || (!AllowNegative &&
Op.getImm() < 0)) {
4626 Twine(
"expected a ") +
4627 (AllowNegative ? Twine(OffsetSize) +
"-bit signed offset"
4628 : Twine(OffsetSize - 1) +
"-bit unsigned offset"));
4637 for (
unsigned i = 2, e =
Operands.size(); i != e; ++i) {
4638 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
4639 if (
Op.isSMEMOffset() ||
Op.isSMEMOffsetMod())
4640 return Op.getStartLoc();
4645bool AMDGPUAsmParser::validateSMEMOffset(
const MCInst &Inst,
4654 auto OpNum = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::offset);
4670 ?
"expected a 23-bit unsigned offset for buffer ops"
4671 :
isGFX12Plus() ?
"expected a 24-bit signed offset"
4672 : (
isVI() || IsBuffer) ?
"expected a 20-bit unsigned offset"
4673 :
"expected a 21-bit signed offset");
4683bool AMDGPUAsmParser::validateBF16InlineConst(
const MCInst &Inst,
4685 if (!getFeatureBits()[AMDGPU::FeatureBF16InlineConstFromUpperFP32])
4700 const int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4704 const MCOperandInfo &Src0Info =
Desc.operands()[Src0Idx];
4709 if (!
Src0.isImm() ||
4711 hasInv2PiInlineImm()))
4716 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
4717 if (ModsIdx != -1 &&
4723 "bf16 inline constant is read from the high half of the fp32 inline "
4724 "constant on this GPU; use the e64 encoding with op_sel:[1,0]");
4728bool AMDGPUAsmParser::validateSOPLiteral(
const MCInst &Inst,
4731 const MCInstrDesc &
Desc = MII.
get(Opcode);
4735 const int Src0Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src0);
4736 const int Src1Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::src1);
4738 const int OpIndices[] = {Src0Idx, Src1Idx};
4740 unsigned NumExprs = 0;
4741 unsigned NumLiterals = 0;
4744 for (
int OpIdx : OpIndices) {
4748 const MCOperand &MO = Inst.
getOperand(OpIdx);
4752 std::optional<int64_t>
Imm;
4755 }
else if (MO.
isExpr()) {
4764 if (!
Imm.has_value()) {
4766 }
else if (!isInlineConstant(Inst, OpIdx)) {
4770 if (NumLiterals == 0 || LiteralValue !=
Value) {
4778 if (NumLiterals + NumExprs <= 1)
4782 "only one unique literal operand is allowed");
4786bool AMDGPUAsmParser::validateOpSel(
const MCInst &Inst) {
4789 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
4797 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
4798 if (OpSelIdx != -1) {
4802 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
4803 if (OpSelHiIdx != -1) {
4812 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
4822 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
4823 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
4824 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
4825 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
4834 auto VerifyOneSGPR = [
OpSel, OpSelHi](
unsigned Index) ->
bool {
4836 return ((OpSel & Mask) == 0) && ((OpSelHi &
Mask) == 0);
4846 int Src2Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2);
4847 if (Src2Idx != -1) {
4858bool AMDGPUAsmParser::validateTrue16OpSel(
const MCInst &Inst) {
4859 if (!hasTrue16Insts())
4861 const MCRegisterInfo *MRI = getMRI();
4863 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
4869 if (OpSelOpValue == 0)
4871 unsigned OpCount = 0;
4872 for (AMDGPU::OpName OpName : {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
4873 AMDGPU::OpName::src2, AMDGPU::OpName::vdst}) {
4874 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), OpName);
4881 bool OpSelOpIsHi = ((OpSelOpValue & (1 << OpCount)) != 0);
4882 if (OpSelOpIsHi != VGPRSuffixIsHi)
4891bool AMDGPUAsmParser::validateNeg(
const MCInst &Inst, AMDGPU::OpName OpName) {
4892 assert(OpName == AMDGPU::OpName::neg_lo || OpName == AMDGPU::OpName::neg_hi);
4904 int NegIdx = AMDGPU::getNamedOperandIdx(
Opc, OpName);
4915 const AMDGPU::OpName SrcMods[3] = {AMDGPU::OpName::src0_modifiers,
4916 AMDGPU::OpName::src1_modifiers,
4917 AMDGPU::OpName::src2_modifiers};
4919 for (
unsigned i = 0; i < 3; ++i) {
4929bool AMDGPUAsmParser::validateDPP(
const MCInst &Inst,
4932 int DppCtrlIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dpp_ctrl);
4933 if (DppCtrlIdx >= 0) {
4937 getSTI().
hasFeature(AMDGPU::FeatureDPALU_DPP) &&
4941 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyDppCtrl,
Operands);
4942 Error(S,
isGFX12() ?
"DP ALU dpp only supports row_share"
4943 :
"DP ALU dpp only supports row_newbcast");
4948 int Dpp8Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::dpp8);
4949 bool IsDPP = DppCtrlIdx >= 0 || Dpp8Idx >= 0;
4952 int Src1Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src1);
4958 "invalid operand for instruction");
4963 "src1 immediate operand invalid for instruction");
4973bool AMDGPUAsmParser::validateVccOperand(MCRegister
Reg)
const {
4974 return (
Reg == AMDGPU::VCC && isWave64()) ||
4975 (
Reg == AMDGPU::VCC_LO && isWave32());
4979bool AMDGPUAsmParser::validateVOPLiteral(
const MCInst &Inst,
4982 const MCInstrDesc &
Desc = MII.
get(Opcode);
4983 bool HasMandatoryLiteral = getNamedOperandIdx(Opcode, OpName::imm) != -1;
4990 std::optional<unsigned> LiteralOpIdx;
4993 for (
int OpIdx : OpIndices) {
4997 const MCOperand &MO = Inst.
getOperand(OpIdx);
5003 std::optional<int64_t>
Imm;
5009 bool IsAnotherLiteral =
false;
5010 bool IsForcedLit = findMCOperand(
Operands, OpIdx).isForcedLit();
5011 bool IsForcedLit64 = findMCOperand(
Operands, OpIdx).isForcedLit64();
5012 if (!
Imm.has_value()) {
5014 IsAnotherLiteral =
true;
5015 }
else if (IsForcedLit || IsForcedLit64 || !isInlineConstant(Inst, OpIdx)) {
5020 HasMandatoryLiteral);
5032 (IsForcedLit64 && !HasMandatoryLiteral)) &&
5033 (!has64BitLiterals() ||
Desc.getSize() != 4)) {
5035 "invalid operand for instruction");
5040 if (!IsForcedFP64 && (IsForcedLit64 || !IsValid32Op) &&
5041 OpIdx != getNamedOperandIdx(Opcode, OpName::src0)) {
5043 "invalid operand for instruction");
5048 if (IsValid32Op && !IsForcedFP64 && !IsForcedLit64) {
5049 Value =
static_cast<uint32_t
>(
5057 if (IsAnotherLiteral && !HasMandatoryLiteral &&
5058 !getFeatureBits()[FeatureVOP3Literal]) {
5060 "literal operands are not supported");
5064 if (LiteralOpIdx && IsAnotherLiteral) {
5066 getOperandLoc(
Operands, *LiteralOpIdx)),
5067 "only one unique literal operand is allowed");
5071 if (IsAnotherLiteral)
5072 LiteralOpIdx = OpIdx;
5081 int OpIdx = AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), Name);
5095bool AMDGPUAsmParser::validateAGPRLdSt(
const MCInst &Inst)
const {
5101 ? AMDGPU::OpName::data0
5102 : AMDGPU::OpName::vdata;
5104 const MCRegisterInfo *MRI = getMRI();
5105 int DstAreg =
IsAGPROperand(Inst, AMDGPU::OpName::vdst, MRI);
5109 int Data2Areg =
IsAGPROperand(Inst, AMDGPU::OpName::data1, MRI);
5110 if (Data2Areg >= 0 && Data2Areg != DataAreg)
5114 auto FB = getFeatureBits();
5115 if (FB[AMDGPU::FeatureGFX90AInsts]) {
5116 if (DataAreg < 0 || DstAreg < 0)
5118 return DstAreg == DataAreg;
5121 return DstAreg < 1 && DataAreg < 1;
5124bool AMDGPUAsmParser::validateVGPRAlign(
const MCInst &Inst)
const {
5125 auto FB = getFeatureBits();
5126 if (!FB[AMDGPU::FeatureRequiresAlignedVGPRs])
5130 const MCRegisterInfo *MRI = getMRI();
5133 if (FB[AMDGPU::FeatureGFX90AInsts] &&
Opc == AMDGPU::DS_READ_B96_TR_B6_vi)
5136 if (FB[AMDGPU::FeatureGFX1250Insts]) {
5140 case AMDGPU::DS_LOAD_TR6_B96:
5141 case AMDGPU::DS_LOAD_TR6_B96_gfx12:
5145 case AMDGPU::GLOBAL_LOAD_TR6_B96:
5146 case AMDGPU::GLOBAL_LOAD_TR6_B96_gfx1250: {
5150 int VAddrIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vaddr);
5151 if (VAddrIdx != -1) {
5154 if ((
Sub - AMDGPU::VGPR0) & 1)
5159 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR:
5160 case AMDGPU::GLOBAL_LOAD_TR6_B96_SADDR_gfx1250:
5165 const MCRegisterClass &VGPR32 = MRI->
getRegClass(AMDGPU::VGPR_32RegClassID);
5166 const MCRegisterClass &AGPR32 = MRI->
getRegClass(AMDGPU::AGPR_32RegClassID);
5186 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
5187 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
5189 return Op.getStartLoc();
5194bool AMDGPUAsmParser::validateBLGP(
const MCInst &Inst,
5197 int BlgpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
5200 SMLoc BLGPLoc = getBLGPLoc(
Operands);
5203 bool IsNeg = StringRef(BLGPLoc.
getPointer()).starts_with(
"neg:");
5204 auto FB = getFeatureBits();
5205 bool UsesNeg =
false;
5206 if (FB[AMDGPU::FeatureGFX940Insts]) {
5208 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_acd:
5209 case AMDGPU::V_MFMA_F64_16X16X4F64_gfx940_vcd:
5210 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_acd:
5211 case AMDGPU::V_MFMA_F64_4X4X4F64_gfx940_vcd:
5216 if (IsNeg == UsesNeg)
5219 Error(BLGPLoc, UsesNeg ?
"invalid modifier: blgp is not supported"
5220 :
"invalid modifier: neg is not supported");
5225bool AMDGPUAsmParser::validateWaitCnt(
const MCInst &Inst,
5231 if (
Opc != AMDGPU::S_WAITCNT_EXPCNT_gfx11 &&
5232 Opc != AMDGPU::S_WAITCNT_LGKMCNT_gfx11 &&
5233 Opc != AMDGPU::S_WAITCNT_VMCNT_gfx11 &&
5234 Opc != AMDGPU::S_WAITCNT_VSCNT_gfx11)
5237 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::sdst);
5240 if (
Reg == AMDGPU::SGPR_NULL)
5243 Error(getOperandLoc(
Operands, Src0Idx),
"src0 must be null");
5247bool AMDGPUAsmParser::validateDS(
const MCInst &Inst,
5252 return validateGWS(Inst,
Operands);
5257 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::gds);
5262 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyGDS,
Operands);
5263 Error(S,
"gds modifier is not supported on this GPU");
5271bool AMDGPUAsmParser::validateGWS(
const MCInst &Inst,
5273 if (!getFeatureBits()[AMDGPU::FeatureGFX90AInsts])
5277 if (
Opc != AMDGPU::DS_GWS_INIT_vi &&
Opc != AMDGPU::DS_GWS_BARRIER_vi &&
5278 Opc != AMDGPU::DS_GWS_SEMA_BR_vi)
5281 const MCRegisterInfo *MRI = getMRI();
5282 const MCRegisterClass &VGPR32 = MRI->
getRegClass(AMDGPU::VGPR_32RegClassID);
5284 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::data0);
5287 auto RegIdx =
Reg - (VGPR32.
contains(
Reg) ? AMDGPU::VGPR0 : AMDGPU::AGPR0);
5289 Error(getOperandLoc(
Operands, Data0Pos),
"vgpr must be even aligned");
5296bool AMDGPUAsmParser::validateCoherencyBits(
const MCInst &Inst,
5300 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::cpol);
5308 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5311 Error(S,
"scale_offset is not supported on this GPU");
5314 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5317 Error(S,
"nv is not supported on this GPU");
5322 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5325 Error(S,
"scale_offset is not supported for this instruction");
5329 return validateTHAndScopeBits(Inst,
Operands, CPol);
5333 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5334 Error(S,
"cache policy is not supported for SMRD instructions");
5338 Error(IDLoc,
"invalid cache policy for SMEM instruction");
5345 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5349 "scc modifier is not supported for this instruction on this GPU");
5360 :
"instruction must use glc");
5365 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5368 &CStr.data()[CStr.find(
isGFX940() ?
"sc0" :
"glc")]);
5370 :
"instruction must not use glc");
5378bool AMDGPUAsmParser::validateTHAndScopeBits(
const MCInst &Inst,
5380 const unsigned CPol) {
5385 SMLoc S = getImmLoc(AMDGPUOperand::ImmTyCPol,
Operands);
5392 return PrintError(
"th:TH_ATOMIC_RETURN requires a destination operand");
5397 return PrintError(
"instruction must use th:TH_ATOMIC_RETURN");
5405 return PrintError(
"invalid th value for SMEM instruction");
5412 return PrintError(
"scope and th combination is not valid");
5418 return PrintError(
"invalid th value for atomic instructions");
5421 return PrintError(
"invalid th value for store instructions");
5424 return PrintError(
"invalid th value for load instructions");
5430bool AMDGPUAsmParser::validateTFE(
const MCInst &Inst,
5434 SMLoc Loc = getImmLoc(AMDGPUOperand::ImmTyTFE,
Operands);
5436 Error(Loc,
"TFE modifier has no meaning for store instructions");
5444bool AMDGPUAsmParser::validateWMMA(
const MCInst &Inst,
5450 int AFmtIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_fmt);
5454 int BFmtIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_fmt);
5457 auto validateFmt = [&](
unsigned Fmt, AMDGPU::OpName SrcOp) ->
bool {
5458 int SrcIdx = AMDGPU::getNamedOperandIdx(
Opc, SrcOp);
5467 "wrong register tuple size for " +
5472 if (!validateFmt(AFmt, AMDGPU::OpName::src0) ||
5473 !validateFmt(BFmt, AMDGPU::OpName::src1))
5477 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale_fmt);
5478 if (AScaleIdx == -1)
5482 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale_fmt);
5486 "invalid matrix and scale format combination");
5493bool AMDGPUAsmParser::validateMonitorSleep(
const MCInst &Inst,
5496 if (
Opc != AMDGPU::S_MONITOR_SLEEP_gfx12 ||
5497 !getSTI().
hasFeature(AMDGPU::FeatureNoSleepForever))
5500 int ImmIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::simm16);
5503 "sleep forever is unsuported on the target");
5510bool AMDGPUAsmParser::validateClusterBarrierIsFirst(
5513 if (
Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx12 &&
5514 Opc != AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM_gfx13)
5517 int Src0Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0);
5524 "s_barrier_signal_isfirst does not support user_cluster_barrier_id (-3)");
5528bool AMDGPUAsmParser::validateScaleSel(
const MCInst &Inst,
5531 int ScaleSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::scale_sel);
5532 if (ScaleSelIdx == -1)
5536 case AMDGPU::V_CVT_SCALE_PK16_F16_FP6_e64_gfx1250:
5537 case AMDGPU::V_CVT_SCALE_PK16_BF16_FP6_e64_gfx1250:
5538 case AMDGPU::V_CVT_SCALE_PK16_F16_BF6_e64_gfx1250:
5539 case AMDGPU::V_CVT_SCALE_PK16_BF16_BF6_e64_gfx1250:
5540 case AMDGPU::V_CVT_SCALE_PK16_F32_FP6_e64_gfx1250:
5541 case AMDGPU::V_CVT_SCALE_PK16_F32_BF6_e64_gfx1250:
5542 case AMDGPU::V_CVT_SCALE_PK8_F16_FP4_e64_gfx1250:
5543 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP4_e64_gfx1250:
5544 case AMDGPU::V_CVT_SCALE_PK8_F32_FP4_e64_gfx1250:
5547 case AMDGPU::V_CVT_SCALE_PK8_F16_FP8_e64_gfx1250:
5548 case AMDGPU::V_CVT_SCALE_PK8_BF16_FP8_e64_gfx1250:
5549 case AMDGPU::V_CVT_SCALE_PK8_F16_BF8_e64_gfx1250:
5550 case AMDGPU::V_CVT_SCALE_PK8_BF16_BF8_e64_gfx1250:
5551 case AMDGPU::V_CVT_SCALE_PK8_F32_FP8_e64_gfx1250:
5552 case AMDGPU::V_CVT_SCALE_PK8_F32_BF8_e64_gfx1250:
5559 if (getSTI().
hasFeature(AMDGPU::FeatureBlock16ConversionScaleInsts))
5563 if (ScaleSel < MaxSel)
5567 "scale_sel maximum supported value is " + Twine(MaxSel - 1));
5571bool AMDGPUAsmParser::validateInstruction(
const MCInst &Inst, SMLoc IDLoc,
5573 if (!validateLdsDirect(Inst,
Operands))
5575 if (!validateTrue16OpSel(Inst)) {
5577 "op_sel operand conflicts with 16-bit operand suffix");
5580 if (!validateSOPLiteral(Inst,
Operands))
5582 if (!validateVOPLiteral(Inst,
Operands)) {
5585 if (!validateConstantBusLimitations(Inst,
Operands)) {
5588 if (!validateVOPD(Inst,
Operands)) {
5591 if (!validateIntClampSupported(Inst)) {
5593 "integer clamping is not supported on this GPU");
5596 if (!validateOpSel(Inst)) {
5598 "invalid op_sel operand");
5601 if (!validateNeg(Inst, AMDGPU::OpName::neg_lo)) {
5603 "invalid neg_lo operand");
5606 if (!validateNeg(Inst, AMDGPU::OpName::neg_hi)) {
5608 "invalid neg_hi operand");
5611 if (!validateDPP(Inst,
Operands)) {
5615 if (!validateMIMGD16(Inst)) {
5617 "d16 modifier is not supported on this GPU");
5620 if (!validateMIMGDim(Inst,
Operands)) {
5621 Error(IDLoc,
"missing dim operand");
5624 if (!validateTensorR128(Inst)) {
5626 "instruction must set modifier r128=0");
5629 if (!validateMIMGMSAA(Inst)) {
5631 "invalid dim; must be MSAA type");
5634 if (!validateMIMGDataSize(Inst, IDLoc)) {
5637 if (!validateMIMGAddrSize(Inst, IDLoc))
5639 if (!validateMIMGAtomicDMask(Inst)) {
5641 "invalid atomic image dmask");
5644 if (!validateMIMGGatherDMask(Inst)) {
5646 "invalid image_gather dmask: only one bit must be set");
5649 if (!validateMovrels(Inst,
Operands)) {
5652 if (!validateOffset(Inst,
Operands)) {
5655 if (!validateBF16InlineConst(Inst,
Operands)) {
5658 if (!validateMAIAccWrite(Inst,
Operands)) {
5661 if (!validateMAISrc2(Inst,
Operands)) {
5664 if (!validateMFMA(Inst,
Operands)) {
5667 if (!validateCoherencyBits(Inst,
Operands, IDLoc)) {
5671 if (!validateAGPRLdSt(Inst)) {
5674 getFeatureBits()[AMDGPU::FeatureGFX90AInsts]
5675 ?
"invalid register class: data and dst should be all VGPR or AGPR"
5676 :
"invalid register class: agpr loads and stores not supported on "
5680 if (!validateVGPRAlign(Inst)) {
5681 Error(IDLoc,
"invalid register class: vgpr tuples must be 64 bit aligned");
5688 if (!validateBLGP(Inst,
Operands)) {
5692 if (!validateDivScale(Inst)) {
5693 Error(IDLoc,
"ABS not allowed in VOP3B instructions");
5696 if (!validateWaitCnt(Inst,
Operands)) {
5699 if (!validateTFE(Inst,
Operands)) {
5702 if (!validateWMMA(Inst,
Operands)) {
5705 if (!validateMonitorSleep(Inst,
Operands)) {
5708 if (!validateClusterBarrierIsFirst(Inst,
Operands)) {
5711 if (!validateScaleSel(Inst,
Operands)) {
5720 unsigned VariantID = 0);
5724 unsigned VariantID);
5726bool AMDGPUAsmParser::isSupportedMnemo(
StringRef Mnemo,
5731bool AMDGPUAsmParser::isSupportedMnemo(StringRef Mnemo,
5732 const FeatureBitset &FBS,
5733 ArrayRef<unsigned> Variants) {
5734 for (
auto Variant : Variants) {
5742bool AMDGPUAsmParser::checkUnsupportedInstruction(StringRef Mnemo,
5744 FeatureBitset FBS = ComputeAvailableFeatures(getFeatureBits());
5747 if (isSupportedMnemo(Mnemo, FBS, getMatchedVariants()))
5752 getParser().clearPendingErrors();
5756 StringRef VariantName = getMatchedVariantName();
5757 if (!VariantName.
empty() && isSupportedMnemo(Mnemo, FBS)) {
5758 return Error(IDLoc, Twine(VariantName,
5759 " variant of this instruction is not supported"));
5763 if (
isGFX10Plus() && getFeatureBits()[AMDGPU::FeatureWavefrontSize64] &&
5764 !getFeatureBits()[AMDGPU::FeatureWavefrontSize32]) {
5766 FeatureBitset FeaturesWS32 = getFeatureBits();
5767 FeaturesWS32.
flip(AMDGPU::FeatureWavefrontSize64)
5768 .
flip(AMDGPU::FeatureWavefrontSize32);
5769 FeatureBitset AvailableFeaturesWS32 =
5770 ComputeAvailableFeatures(FeaturesWS32);
5772 if (isSupportedMnemo(Mnemo, AvailableFeaturesWS32, getMatchedVariants()))
5773 return Error(IDLoc,
"instruction requires wavesize=32");
5777 if (isSupportedMnemo(Mnemo, FeatureBitset().set())) {
5778 return Error(IDLoc,
"instruction not supported on this GPU (" +
5779 getSTI().
getCPU() +
")" +
": " + Mnemo);
5784 return Error(IDLoc,
"invalid instruction" + Suggestion);
5790 const auto &
Op = ((AMDGPUOperand &)*
Operands[InvalidOprIdx]);
5791 if (
Op.isToken() && InvalidOprIdx > 1) {
5792 const auto &PrevOp = ((AMDGPUOperand &)*
Operands[InvalidOprIdx - 1]);
5793 return PrevOp.isToken() && PrevOp.getToken() ==
"::";
5798bool AMDGPUAsmParser::matchAndEmitInstruction(SMLoc IDLoc,
unsigned &Opcode,
5802 bool MatchingInlineAsm) {
5805 unsigned Result = Match_Success;
5810 auto atLeastAsSpecific = [](
unsigned New,
unsigned Cur) {
5811 auto rank = [](
unsigned M) {
5812 return M == Match_MnemonicFail ? 1
5813 :
M == Match_InvalidOperand ? 2
5814 :
M == Match_MissingFeature ? 3
5817 return rank(New) >= rank(Cur);
5820 for (
auto Variant : getMatchedVariants()) {
5823 MatchInstructionImpl(
Operands, Inst, EI, MatchingInlineAsm, Variant);
5824 if (R == Match_Success || atLeastAsSpecific(R, Result)) {
5828 if (R == Match_Success)
5832 if (Result == Match_Success) {
5833 if (!validateInstruction(Inst, IDLoc,
Operands)) {
5836 emitTargetDirective();
5837 Out.emitInstruction(Inst, getSTI());
5844 if (checkUnsupportedInstruction(Mnemo, IDLoc)) {
5851 case Match_MissingFeature:
5855 return Error(IDLoc,
"operands are not valid for this GPU or mode");
5857 case Match_InvalidOperand: {
5858 SMLoc ErrorLoc = IDLoc;
5859 if (ErrorInfo != ~0ULL) {
5860 if (ErrorInfo >=
Operands.size()) {
5861 return Error(IDLoc,
"too few operands for instruction");
5863 AMDGPUOperand &ErrorOp = (AMDGPUOperand &)*
Operands[ErrorInfo];
5864 ErrorLoc = ErrorOp.getStartLoc();
5865 if (ErrorLoc == SMLoc())
5869 return Error(ErrorLoc,
"invalid VOPDY instruction");
5871 return Error(ErrorLoc,
"invalid operand for instruction");
5874 case Match_MnemonicFail:
5880bool AMDGPUAsmParser::ParseAsAbsoluteExpression(uint32_t &Ret) {
5885 if (getParser().parseAbsoluteExpression(Tmp)) {
5888 Ret =
static_cast<uint32_t
>(Tmp);
5892bool AMDGPUAsmParser::ParseDirectiveAMDGCNTarget() {
5893 if (!getSTI().getTargetTriple().isAMDGCN())
5894 return TokError(
"directive only supported for amdgcn architecture");
5896 std::string TargetIDDirective;
5897 SMLoc TargetStart = getTok().getLoc();
5898 if (getParser().parseEscapedString(TargetIDDirective))
5901 std::optional<AMDGPU::TargetID> MaybeParsed =
5904 return getParser().Error(TargetStart,
5905 "malformed target id '" + TargetIDDirective +
"'");
5908 const Triple &
TT = getSTI().getTargetTriple();
5914 return getParser().Error(
5915 TargetStart,
"target id '" + TargetIDDirective +
5916 "' specifies a processor that is not valid for "
5918 TT.getArchName() +
"'");
5921 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
5922 getTargetStreamer().getTargetID();
5925 const Triple &STITriple = getSTI().getTargetTriple();
5926 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
5927 return getParser().Error(
5928 TargetStart,
".amdgcn_target " + Twine(ParsedTargetID.
toString()) +
5929 " is incompatible with " +
5930 Twine(CurrentTargetID->toString()));
5934 StringRef DirectiveProcessor =
5937 if (DirectiveISA != ISA) {
5938 return getParser().Error(TargetStart,
5939 ".amdgcn_target directive processor " +
5940 Twine(DirectiveProcessor) +
5941 " does not match the specified processor " +
5942 Twine(getSTI().
getCPU()));
5948 CurrentTargetID->getXnackSetting())) {
5950 ".amdgcn_target directive has conflicting xnack settings");
5954 CurrentTargetID->getSramEccSetting())) {
5956 ".amdgcn_target directive has conflicting sramecc settings");
5962 getTargetStreamer().getTargetID()->setXnackSetting(
5964 getTargetStreamer().getTargetID()->setSramEccSetting(
5970bool AMDGPUAsmParser::OutOfRangeError(SMRange
Range) {
5974bool AMDGPUAsmParser::calculateGPRBlocks(
5975 const FeatureBitset &Features,
const MCExpr *VCCUsed,
5976 const MCExpr *FlatScrUsed,
bool XNACKUsed,
5977 std::optional<bool> EnableWavefrontSize32,
const MCExpr *NextFreeVGPR,
5978 SMRange VGPRRange,
const MCExpr *NextFreeSGPR, SMRange SGPRRange,
5979 const MCExpr *&VGPRBlocks,
const MCExpr *&SGPRBlocks) {
5984 const MCExpr *
NumSGPRs = NextFreeSGPR;
5985 int64_t EvaluatedSGPRs;
5987 if (
ISA.Major >= 10)
5992 if (
NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
ISA.Major >= 8 &&
5993 !Features.
test(FeatureSGPRInitBug) &&
5994 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
5995 return OutOfRangeError(SGPRRange);
5997 const MCExpr *ExtraSGPRs =
6001 if (
NumSGPRs->evaluateAsAbsolute(EvaluatedSGPRs) &&
6002 (
ISA.Major <= 7 || Features.
test(FeatureSGPRInitBug)) &&
6003 static_cast<uint64_t>(EvaluatedSGPRs) > MaxAddressableNumSGPRs)
6004 return OutOfRangeError(SGPRRange);
6006 if (Features.
test(FeatureSGPRInitBug))
6013 auto GetNumGPRBlocks = [&Ctx](
const MCExpr *NumGPR,
6014 unsigned Granule) ->
const MCExpr * {
6018 const MCExpr *AlignToGPR =
6020 const MCExpr *DivGPR =
6026 VGPRBlocks = GetNumGPRBlocks(
6035bool AMDGPUAsmParser::ParseDirectiveAMDHSAKernel() {
6036 if (!getSTI().getTargetTriple().isAMDGCN())
6037 return TokError(
"directive only supported for amdgcn architecture");
6040 return TokError(
"directive only supported for amdhsa OS");
6042 StringRef KernelName;
6043 if (getParser().parseIdentifier(KernelName))
6050 AMDGPU::MCKernelDescriptor KD =
6060 const MCExpr *NextFreeVGPR = ZeroExpr;
6062 const MCExpr *NamedBarCnt = ZeroExpr;
6067 const MCExpr *NextFreeSGPR = ZeroExpr;
6070 unsigned ImpliedUserSGPRCount = 0;
6074 std::optional<unsigned> ExplicitUserSGPRCount;
6075 const MCExpr *ReserveVCC = OneExpr;
6076 const MCExpr *ReserveFlatScr = OneExpr;
6077 std::optional<bool> EnableWavefrontSize32;
6084 SMRange IDRange = getTok().getLocRange();
6085 if (!parseId(ID,
"expected .amdhsa_ directive or .end_amdhsa_kernel"))
6088 if (ID ==
".end_amdhsa_kernel")
6091 if (!Seen.
insert(ID).second)
6092 return TokError(
".amdhsa_ directives cannot be repeated");
6094 SMLoc ValStart = getLoc();
6095 const MCExpr *ExprVal;
6096 if (getParser().parseExpression(ExprVal))
6098 SMLoc ValEnd = getLoc();
6099 SMRange ValRange = SMRange(ValStart, ValEnd);
6103 bool EvaluatableExpr;
6104 if ((EvaluatableExpr = ExprVal->evaluateAsAbsolute(IVal))) {
6106 return OutOfRangeError(ValRange);
6110#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE) \
6111 if (!isUInt<ENTRY##_WIDTH>(Val)) \
6112 return OutOfRangeError(RANGE); \
6113 AMDGPU::MCKernelDescriptor::bits_set(FIELD, VALUE, ENTRY##_SHIFT, ENTRY, \
6118#define EXPR_RESOLVE_OR_ERROR(RESOLVED) \
6120 return Error(IDRange.Start, "directive should have resolvable expression", \
6123 if (ID ==
".amdhsa_group_segment_fixed_size") {
6126 return OutOfRangeError(ValRange);
6128 }
else if (ID ==
".amdhsa_private_segment_fixed_size") {
6131 return OutOfRangeError(ValRange);
6133 }
else if (ID ==
".amdhsa_kernarg_size") {
6135 return OutOfRangeError(ValRange);
6137 }
else if (ID ==
".amdhsa_user_sgpr_count") {
6139 ExplicitUserSGPRCount = Val;
6140 }
else if (ID ==
".amdhsa_user_sgpr_private_segment_buffer") {
6144 "directive is not supported with architected flat scratch",
6147 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_BUFFER,
6150 ImpliedUserSGPRCount += 4;
6151 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_preload_length") {
6154 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6157 return OutOfRangeError(ValRange);
6161 ImpliedUserSGPRCount += Val;
6162 PreloadLength = Val;
6164 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_preload_offset") {
6167 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6170 return OutOfRangeError(ValRange);
6174 PreloadOffset = Val;
6175 }
else if (ID ==
".amdhsa_user_sgpr_dispatch_ptr") {
6178 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_PTR, ExprVal,
6181 ImpliedUserSGPRCount += 2;
6182 }
else if (ID ==
".amdhsa_user_sgpr_queue_ptr") {
6185 KERNEL_CODE_PROPERTY_ENABLE_SGPR_QUEUE_PTR, ExprVal,
6188 ImpliedUserSGPRCount += 2;
6189 }
else if (ID ==
".amdhsa_user_sgpr_kernarg_segment_ptr") {
6192 KERNEL_CODE_PROPERTY_ENABLE_SGPR_KERNARG_SEGMENT_PTR,
6195 ImpliedUserSGPRCount += 2;
6196 }
else if (ID ==
".amdhsa_user_sgpr_dispatch_id") {
6199 KERNEL_CODE_PROPERTY_ENABLE_SGPR_DISPATCH_ID, ExprVal,
6202 ImpliedUserSGPRCount += 2;
6203 }
else if (ID ==
".amdhsa_user_sgpr_flat_scratch_init") {
6206 "directive is not supported with architected flat scratch",
6210 KERNEL_CODE_PROPERTY_ENABLE_SGPR_FLAT_SCRATCH_INIT,
6213 ImpliedUserSGPRCount += 2;
6214 }
else if (ID ==
".amdhsa_user_sgpr_private_segment_size") {
6217 KERNEL_CODE_PROPERTY_ENABLE_SGPR_PRIVATE_SEGMENT_SIZE,
6220 ImpliedUserSGPRCount += 1;
6221 }
else if (ID ==
".amdhsa_wavefront_size32") {
6224 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6225 EnableWavefrontSize32 = Val;
6227 KERNEL_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32, ExprVal,
6229 }
else if (ID ==
".amdhsa_uses_dynamic_stack") {
6231 KERNEL_CODE_PROPERTY_USES_DYNAMIC_STACK, ExprVal,
6233 }
else if (ID ==
".amdhsa_system_sgpr_private_segment_wavefront_offset") {
6236 "directive is not supported with architected flat scratch",
6239 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6241 }
else if (ID ==
".amdhsa_enable_private_segment") {
6245 "directive is not supported without architected flat scratch",
6248 COMPUTE_PGM_RSRC2_ENABLE_PRIVATE_SEGMENT, ExprVal,
6250 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_x") {
6252 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_X, ExprVal,
6254 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_y") {
6256 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Y, ExprVal,
6258 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_id_z") {
6260 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_ID_Z, ExprVal,
6262 }
else if (ID ==
".amdhsa_system_sgpr_workgroup_info") {
6264 COMPUTE_PGM_RSRC2_ENABLE_SGPR_WORKGROUP_INFO, ExprVal,
6266 }
else if (ID ==
".amdhsa_system_vgpr_workitem_id") {
6268 COMPUTE_PGM_RSRC2_ENABLE_VGPR_WORKITEM_ID, ExprVal,
6270 }
else if (ID ==
".amdhsa_next_free_vgpr") {
6271 VGPRRange = ValRange;
6272 NextFreeVGPR = ExprVal;
6273 }
else if (ID ==
".amdhsa_next_free_sgpr") {
6274 SGPRRange = ValRange;
6275 NextFreeSGPR = ExprVal;
6276 }
else if (ID ==
".amdhsa_accum_offset") {
6278 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6279 AccumOffset = ExprVal;
6280 }
else if (ID ==
".amdhsa_named_barrier_count") {
6282 return Error(IDRange.
Start,
"directive requires gfx1250+", IDRange);
6283 NamedBarCnt = ExprVal;
6284 }
else if (ID ==
".amdhsa_reserve_vcc") {
6286 return OutOfRangeError(ValRange);
6287 ReserveVCC = ExprVal;
6288 }
else if (ID ==
".amdhsa_reserve_flat_scratch") {
6290 return Error(IDRange.
Start,
"directive requires gfx7+", IDRange);
6293 "directive is not supported with architected flat scratch",
6296 return OutOfRangeError(ValRange);
6297 ReserveFlatScr = ExprVal;
6298 }
else if (ID ==
".amdhsa_reserve_xnack_mask") {
6300 return Error(IDRange.
Start,
"directive requires gfx8+", IDRange);
6302 return OutOfRangeError(ValRange);
6303 bool XnackOn = getTargetStreamer().getTargetID()->isXnackOnOrAny();
6304 if (Val != XnackOn) {
6305 return getParser().Error(
6307 ".amdhsa_reserve_xnack_mask does not match target id", IDRange);
6309 }
else if (ID ==
".amdhsa_float_round_mode_32") {
6311 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_32, ExprVal,
6313 }
else if (ID ==
".amdhsa_float_round_mode_16_64") {
6315 COMPUTE_PGM_RSRC1_FLOAT_ROUND_MODE_16_64, ExprVal,
6317 }
else if (ID ==
".amdhsa_float_denorm_mode_32") {
6319 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_32, ExprVal,
6321 }
else if (ID ==
".amdhsa_float_denorm_mode_16_64") {
6323 COMPUTE_PGM_RSRC1_FLOAT_DENORM_MODE_16_64, ExprVal,
6325 }
else if (ID ==
".amdhsa_dx10_clamp") {
6326 if (!getSTI().
hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6327 return Error(IDRange.
Start,
"directive unsupported on gfx1170+",
6330 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_DX10_CLAMP, ExprVal,
6332 }
else if (ID ==
".amdhsa_ieee_mode") {
6333 if (!getSTI().
hasFeature(AMDGPU::FeatureDX10ClampAndIEEEMode))
6334 return Error(IDRange.
Start,
"directive unsupported on gfx1170+",
6337 COMPUTE_PGM_RSRC1_GFX6_GFX11_ENABLE_IEEE_MODE, ExprVal,
6339 }
else if (ID ==
".amdhsa_fp16_overflow") {
6341 return Error(IDRange.
Start,
"directive requires gfx9+", IDRange);
6343 COMPUTE_PGM_RSRC1_GFX9_PLUS_FP16_OVFL, ExprVal,
6345 }
else if (ID ==
".amdhsa_tg_split") {
6347 return Error(IDRange.
Start,
"directive requires gfx90a+", IDRange);
6350 }
else if (ID ==
".amdhsa_workgroup_processor_mode") {
6353 "directive unsupported on " + getSTI().
getCPU(), IDRange);
6355 COMPUTE_PGM_RSRC1_GFX10_PLUS_WGP_MODE, ExprVal,
6357 }
else if (ID ==
".amdhsa_memory_ordered") {
6359 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6361 COMPUTE_PGM_RSRC1_GFX10_PLUS_MEM_ORDERED, ExprVal,
6363 }
else if (ID ==
".amdhsa_forward_progress") {
6365 return Error(IDRange.
Start,
"directive requires gfx10+", IDRange);
6367 COMPUTE_PGM_RSRC1_GFX10_PLUS_FWD_PROGRESS, ExprVal,
6369 }
else if (ID ==
".amdhsa_shared_vgpr_count") {
6371 if (
ISA.Major < 10 ||
ISA.Major >= 12)
6372 return Error(IDRange.
Start,
"directive requires gfx10 or gfx11",
6374 SharedVGPRCount = Val;
6376 COMPUTE_PGM_RSRC3_GFX10_GFX11_SHARED_VGPR_COUNT, ExprVal,
6378 }
else if (ID ==
".amdhsa_inst_pref_size") {
6380 return Error(IDRange.
Start,
"directive requires gfx11+", IDRange);
6381 if (
ISA.Major == 11) {
6383 COMPUTE_PGM_RSRC3_GFX11_INST_PREF_SIZE, ExprVal,
6387 COMPUTE_PGM_RSRC3_GFX12_PLUS_INST_PREF_SIZE, ExprVal,
6390 }
else if (ID ==
".amdhsa_exception_fp_ieee_invalid_op") {
6393 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INVALID_OPERATION,
6395 }
else if (ID ==
".amdhsa_exception_fp_denorm_src") {
6397 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_FP_DENORMAL_SOURCE,
6399 }
else if (ID ==
".amdhsa_exception_fp_ieee_div_zero") {
6402 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_DIVISION_BY_ZERO,
6404 }
else if (ID ==
".amdhsa_exception_fp_ieee_overflow") {
6406 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_OVERFLOW,
6408 }
else if (ID ==
".amdhsa_exception_fp_ieee_underflow") {
6410 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_UNDERFLOW,
6412 }
else if (ID ==
".amdhsa_exception_fp_ieee_inexact") {
6414 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_IEEE_754_FP_INEXACT,
6416 }
else if (ID ==
".amdhsa_exception_int_div_zero") {
6418 COMPUTE_PGM_RSRC2_ENABLE_EXCEPTION_INT_DIVIDE_BY_ZERO,
6420 }
else if (ID ==
".amdhsa_round_robin_scheduling") {
6422 return Error(IDRange.
Start,
"directive requires gfx12+", IDRange);
6424 COMPUTE_PGM_RSRC1_GFX12_PLUS_ENABLE_WG_RR_EN, ExprVal,
6427 return Error(IDRange.
Start,
"unknown .amdhsa_kernel directive", IDRange);
6430#undef PARSE_BITS_ENTRY
6433 if (!Seen.
contains(
".amdhsa_next_free_vgpr"))
6434 return TokError(
".amdhsa_next_free_vgpr directive is required");
6436 if (!Seen.
contains(
".amdhsa_next_free_sgpr"))
6437 return TokError(
".amdhsa_next_free_sgpr directive is required");
6439 unsigned UserSGPRCount = ExplicitUserSGPRCount.value_or(ImpliedUserSGPRCount);
6441 return TokError(
"too many user SGPRs enabled, found " +
6442 Twine(UserSGPRCount) +
", but only " +
6448 if (PreloadLength) {
6454 const MCExpr *VGPRBlocks;
6455 const MCExpr *SGPRBlocks;
6456 if (calculateGPRBlocks(getFeatureBits(), ReserveVCC, ReserveFlatScr,
6457 getTargetStreamer().getTargetID()->isXnackOnOrAny(),
6458 EnableWavefrontSize32, NextFreeVGPR, VGPRRange,
6459 NextFreeSGPR, SGPRRange, VGPRBlocks, SGPRBlocks))
6462 int64_t EvaluatedVGPRBlocks;
6463 bool VGPRBlocksEvaluatable =
6464 VGPRBlocks->evaluateAsAbsolute(EvaluatedVGPRBlocks);
6465 if (VGPRBlocksEvaluatable &&
6467 static_cast<uint64_t>(EvaluatedVGPRBlocks))) {
6468 return OutOfRangeError(VGPRRange);
6472 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT_SHIFT,
6473 COMPUTE_PGM_RSRC1_GRANULATED_WORKITEM_VGPR_COUNT,
getContext());
6475 int64_t EvaluatedSGPRBlocks;
6476 if (SGPRBlocks->evaluateAsAbsolute(EvaluatedSGPRBlocks) &&
6478 static_cast<uint64_t>(EvaluatedSGPRBlocks)))
6479 return OutOfRangeError(SGPRRange);
6482 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT_SHIFT,
6483 COMPUTE_PGM_RSRC1_GRANULATED_WAVEFRONT_SGPR_COUNT,
getContext());
6485 if (ExplicitUserSGPRCount && ImpliedUserSGPRCount > *ExplicitUserSGPRCount)
6486 return TokError(
"amdgpu_user_sgpr_count smaller than implied by "
6487 "enabled user SGPRs");
6493 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT_SHIFT,
6494 COMPUTE_PGM_RSRC2_GFX125_USER_SGPR_COUNT,
getContext());
6499 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT_SHIFT,
6500 COMPUTE_PGM_RSRC2_GFX6_GFX120_USER_SGPR_COUNT,
getContext());
6505 return TokError(
"Kernarg size should be resolvable");
6507 if (PreloadLength && kernarg_size &&
6508 (PreloadLength * 4 + PreloadOffset * 4 > kernarg_size))
6509 return TokError(
"Kernarg preload length + offset is larger than the "
6510 "kernarg segment size");
6513 if (!Seen.
contains(
".amdhsa_accum_offset"))
6514 return TokError(
".amdhsa_accum_offset directive is required");
6515 int64_t EvaluatedAccum;
6516 bool AccumEvaluatable = AccumOffset->evaluateAsAbsolute(EvaluatedAccum);
6517 uint64_t UEvaluatedAccum = EvaluatedAccum;
6518 if (AccumEvaluatable &&
6519 (UEvaluatedAccum < 4 || UEvaluatedAccum > 256 || (UEvaluatedAccum & 3)))
6520 return TokError(
"accum_offset should be in range [4..256] in "
6523 int64_t EvaluatedNumVGPR;
6524 if (NextFreeVGPR->evaluateAsAbsolute(EvaluatedNumVGPR) &&
6528 return TokError(
"accum_offset exceeds total VGPR allocation");
6534 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET_SHIFT,
6535 COMPUTE_PGM_RSRC3_GFX90A_ACCUM_OFFSET,
6541 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT_SHIFT,
6542 COMPUTE_PGM_RSRC3_GFX125_NAMED_BAR_CNT,
6545 if (
ISA.Major >= 10 &&
ISA.Major < 12) {
6547 if (SharedVGPRCount && EnableWavefrontSize32 && *EnableWavefrontSize32) {
6548 return TokError(
"shared_vgpr_count directive not valid on "
6549 "wavefront size 32");
6552 if (VGPRBlocksEvaluatable &&
6553 (SharedVGPRCount * 2 +
static_cast<uint64_t>(EvaluatedVGPRBlocks) >
6555 return TokError(
"shared_vgpr_count*2 + "
6556 "compute_pgm_rsrc1.GRANULATED_WORKITEM_VGPR_COUNT cannot "
6561 emitTargetDirective();
6562 getTargetStreamer().EmitAmdhsaKernelDescriptor(getSTI(), KernelName, KD,
6563 NextFreeVGPR, NextFreeSGPR,
6564 ReserveVCC, ReserveFlatScr);
6568bool AMDGPUAsmParser::ParseDirectiveAMDHSACodeObjectVersion() {
6570 if (ParseAsAbsoluteExpression(
Version))
6573 getTargetStreamer().EmitDirectiveAMDHSACodeObjectVersion(
Version);
6574 emitTargetDirective();
6578bool AMDGPUAsmParser::ParseAMDKernelCodeTValue(StringRef ID,
6579 AMDGPUMCKernelCodeT &
C) {
6582 if (ID ==
"max_scratch_backing_memory_byte_size") {
6583 Parser.eatToEndOfStatement();
6587 SmallString<40> ErrStr;
6588 raw_svector_ostream Err(ErrStr);
6589 if (!
C.ParseKernelCodeT(ID, getParser(), Err)) {
6590 return TokError(Err.
str());
6594 if (ID ==
"enable_wavefront_size32") {
6597 return TokError(
"enable_wavefront_size32=1 is only allowed on GFX10+");
6599 return TokError(
"enable_wavefront_size32=1 requires +WavefrontSize32");
6602 return TokError(
"enable_wavefront_size32=0 requires +WavefrontSize64");
6606 if (ID ==
"wavefront_size") {
6607 if (
C.wavefront_size == 5) {
6609 return TokError(
"wavefront_size=5 is only allowed on GFX10+");
6611 return TokError(
"wavefront_size=5 requires +WavefrontSize32");
6612 }
else if (
C.wavefront_size == 6) {
6614 return TokError(
"wavefront_size=6 requires +WavefrontSize64");
6621bool AMDGPUAsmParser::ParseDirectiveAMDKernelCodeT() {
6622 AMDGPUMCKernelCodeT KernelCode;
6632 if (!parseId(ID,
"expected value identifier or .end_amd_kernel_code_t"))
6635 if (ID ==
".end_amd_kernel_code_t")
6638 if (ParseAMDKernelCodeTValue(ID, KernelCode))
6643 getTargetStreamer().EmitAMDKernelCodeT(KernelCode);
6648bool AMDGPUAsmParser::ParseDirectiveAMDGPUHsaKernel() {
6649 StringRef KernelName;
6650 if (!parseId(KernelName,
"expected symbol name"))
6653 getTargetStreamer().EmitAMDGPUSymbolType(KernelName,
6660bool AMDGPUAsmParser::ParseDirectiveISAVersion() {
6661 if (!getSTI().getTargetTriple().isAMDGCN()) {
6662 return Error(getLoc(),
6663 ".amd_amdgpu_isa directive is not available on non-amdgcn "
6667 StringRef TargetIDDirective = getLexer().getTok().getStringContents();
6669 std::optional<AMDGPU::TargetID> MaybeParsed =
6672 return Error(getParser().getTok().getLoc(),
6673 "malformed target id '" + TargetIDDirective +
"'");
6676 const Triple &
TT = getSTI().getTargetTriple();
6682 return Error(getParser().getTok().getLoc(),
6683 "target id '" + TargetIDDirective +
6684 "' specifies a processor that is not valid for subarch '" +
6685 TT.getArchName() +
"'");
6688 const std::optional<AMDGPU::TargetID> &CurrentTargetID =
6689 getTargetStreamer().getTargetID();
6692 const Triple &STITriple = getSTI().getTargetTriple();
6693 if (!DirectiveTriple.isCompatibleWith(STITriple)) {
6694 return Error(getParser().getTok().getLoc(),
6695 ".amd_amdgpu_isa " + Twine(ParsedTargetID.
toString()) +
6696 " is incompatible with " +
6697 Twine(CurrentTargetID->toString()));
6701 StringRef DirectiveProcessor =
6704 if (DirectiveISA != ISA) {
6705 return Error(getParser().getTok().getLoc(),
6706 ".amd_amdgpu_isa directive processor " +
6707 Twine(DirectiveProcessor) +
6708 " does not match the specified processor " +
6709 Twine(getSTI().
getCPU()));
6712 getTargetStreamer().EmitISAVersion();
6718bool AMDGPUAsmParser::ParseDirectiveHSAMetadata() {
6721 std::string HSAMetadataString;
6726 if (!getTargetStreamer().EmitHSAMetadataV3(HSAMetadataString))
6727 return Error(getLoc(),
"invalid HSA metadata");
6734bool AMDGPUAsmParser::ParseToEndDirective(
const char *AssemblerDirectiveBegin,
6735 const char *AssemblerDirectiveEnd,
6736 std::string &CollectString) {
6738 raw_string_ostream CollectStream(CollectString);
6740 getLexer().setSkipSpace(
false);
6742 bool FoundEnd =
false;
6745 CollectStream << getTokenStr();
6749 if (trySkipId(AssemblerDirectiveEnd)) {
6754 CollectStream << Parser.parseStringToEndOfStatement()
6755 <<
getContext().getAsmInfo().getSeparatorString();
6757 Parser.eatToEndOfStatement();
6760 getLexer().setSkipSpace(
true);
6763 return TokError(Twine(
"expected directive ") +
6764 Twine(AssemblerDirectiveEnd) + Twine(
" not found"));
6771bool AMDGPUAsmParser::ParseDirectivePALMetadataBegin() {
6777 auto *PALMetadata = getTargetStreamer().getPALMetadata();
6778 if (!PALMetadata->setFromString(
String))
6779 return Error(getLoc(),
"invalid PAL metadata");
6784bool AMDGPUAsmParser::ParseDirectivePALMetadata() {
6787 Twine(
" directive is "
6788 "not available on non-amdpal OSes"))
6792 auto *PALMetadata = getTargetStreamer().getPALMetadata();
6793 PALMetadata->setLegacy();
6796 if (ParseAsAbsoluteExpression(
Key)) {
6797 return TokError(Twine(
"invalid value in ") +
6801 return TokError(Twine(
"expected an even number of values in ") +
6804 if (ParseAsAbsoluteExpression(
Value)) {
6805 return TokError(Twine(
"invalid value in ") +
6808 PALMetadata->setRegister(
Key,
Value);
6817bool AMDGPUAsmParser::ParseDirectiveAMDGPULDS() {
6818 if (getParser().checkForValidSection())
6822 SMLoc NameLoc = getLoc();
6823 if (getParser().parseIdentifier(Name))
6824 return TokError(
"expected identifier in directive");
6827 if (getParser().parseComma())
6833 SMLoc SizeLoc = getLoc();
6834 if (getParser().parseAbsoluteExpression(
Size))
6837 return Error(SizeLoc,
"size must be non-negative");
6838 if (
Size > LocalMemorySize)
6839 return Error(SizeLoc,
"size is too large");
6843 SMLoc AlignLoc = getLoc();
6844 if (getParser().parseAbsoluteExpression(Alignment))
6847 return Error(AlignLoc,
"alignment must be a power of two");
6852 if (Alignment >= 1u << 31)
6853 return Error(AlignLoc,
"alignment is too large");
6859 Symbol->redefineIfPossible();
6860 if (!
Symbol->isUndefined())
6861 return Error(NameLoc,
"invalid symbol redefinition");
6863 getTargetStreamer().emitAMDGPULDS(Symbol,
Size,
Align(Alignment));
6867bool AMDGPUAsmParser::ParseDirectiveAMDGPUInfo() {
6868 if (getParser().checkForValidSection())
6872 if (getParser().parseIdentifier(FuncName))
6873 return TokError(
"expected symbol name after .amdgpu_info");
6876 AMDGPU::InfoSectionData ParsedInfoData;
6877 AMDGPU::FuncInfo FI;
6879 bool HasScalarAttrs =
false;
6886 SMLoc IDLoc = getLoc();
6887 if (!parseId(ID,
"expected directive or .end_amdgpu_info"))
6890 if (ID ==
".end_amdgpu_info")
6898 return Error(IDLoc,
"unknown .amdgpu_info directive '" + ID +
"'");
6900 if (Dir ==
"flags") {
6902 if (getParser().parseAbsoluteExpression(Val))
6905 FI.
UsesVCC = !!(
Flags & AMDGPU::FuncInfoFlags::FUNC_USES_VCC);
6907 !!(
Flags & AMDGPU::FuncInfoFlags::FUNC_USES_FLAT_SCRATCH);
6909 HasScalarAttrs =
true;
6910 }
else if (Dir ==
"num_sgpr") {
6912 if (getParser().parseAbsoluteExpression(Val))
6914 FI.
NumSGPR =
static_cast<uint32_t
>(Val);
6915 HasScalarAttrs =
true;
6916 }
else if (Dir ==
"num_vgpr") {
6918 if (getParser().parseAbsoluteExpression(Val))
6921 HasScalarAttrs =
true;
6922 }
else if (Dir ==
"num_agpr") {
6924 if (getParser().parseAbsoluteExpression(Val))
6927 HasScalarAttrs =
true;
6928 }
else if (Dir ==
"private_segment_size") {
6930 if (getParser().parseAbsoluteExpression(Val))
6933 HasScalarAttrs =
true;
6934 }
else if (Dir ==
"use") {
6936 if (getParser().parseIdentifier(ResName))
6937 return TokError(
"expected resource symbol for .amdgpu_use");
6938 ParsedInfoData.
Uses.push_back(
6939 {FuncSym,
getContext().getOrCreateSymbol(ResName)});
6940 }
else if (Dir ==
"call") {
6942 if (getParser().parseIdentifier(DstName))
6943 return TokError(
"expected callee symbol for .amdgpu_call");
6944 ParsedInfoData.
Calls.push_back(
6945 {FuncSym,
getContext().getOrCreateSymbol(DstName)});
6946 }
else if (Dir ==
"indirect_call") {
6948 if (getParser().parseEscapedString(TypeId))
6949 return TokError(
"expected type ID string for .amdgpu_indirect_call");
6950 ParsedInfoData.
IndirectCalls.push_back({FuncSym, std::move(TypeId)});
6951 }
else if (Dir ==
"typeid") {
6953 if (getParser().parseEscapedString(TypeId))
6954 return TokError(
"expected type ID string for .amdgpu_typeid");
6955 ParsedInfoData.
TypeIds.push_back({FuncSym, std::move(TypeId)});
6957 return Error(IDLoc,
"unknown .amdgpu_info directive '" + ID +
"'");
6962 ParsedInfoData.
Funcs.push_back(std::move(FI));
6964 AMDGPU::InfoSectionData &
Data = InfoData ? *InfoData : InfoData.emplace();
6965 for (AMDGPU::FuncInfo &Func : ParsedInfoData.
Funcs)
6966 Data.Funcs.push_back(std::move(Func));
6967 for (std::pair<MCSymbol *, MCSymbol *> &Use : ParsedInfoData.
Uses)
6968 Data.Uses.push_back(Use);
6969 for (std::pair<MCSymbol *, MCSymbol *> &
Call : ParsedInfoData.
Calls)
6971 for (std::pair<MCSymbol *, std::string> &
IndirectCall :
6974 for (std::pair<MCSymbol *, std::string> &TypeId : ParsedInfoData.
TypeIds)
6975 Data.TypeIds.push_back(std::move(TypeId));
6980void AMDGPUAsmParser::doBeforeLabelEmit(MCSymbol *Symbol, SMLoc IDLoc) {
6987void AMDGPUAsmParser::checkKernelPrologues() {
6988 if (getFeatureBits()[AMDGPU::FeatureRequiresInitialUnclausedVmem]) {
6989 static const unsigned Required[] = {S_MOV_B64_gfx12, V_NOP_e32_gfx12,
6990 GLOBAL_PREFETCH_B8_SADDR_gfx1250};
6991 for (
auto [Sym, Loc,
Offset] : OpcodeStreamSymbols) {
6992 if (!AMDHSAKernelSymbols.
contains(Sym))
6994 ArrayRef<unsigned> Prologue =
ArrayRef(OpcodeStream).drop_front(
Offset);
6995 if (!Prologue.
empty() && Prologue.
front() == S_SETREG_IMM32_B32_gfx12)
6999 "' does not begin with the required prologue "
7000 "sequence: s_mov_b64 followed by v_nop and "
7001 "global_prefetch_b8");
7005 OpcodeStream.
clear();
7006 OpcodeStreamSymbols.clear();
7007 AMDHSAKernelSymbols.
clear();
7010void AMDGPUAsmParser::onEndOfFile() {
7011 emitTargetDirective();
7012 checkKernelPrologues();
7014 getTargetStreamer().emitAMDGPUInfo(*InfoData);
7017bool AMDGPUAsmParser::ParseDirective(AsmToken DirectiveID) {
7018 StringRef IDVal = DirectiveID.
getString();
7021 if (IDVal ==
".amdhsa_kernel")
7022 return ParseDirectiveAMDHSAKernel();
7024 if (IDVal ==
".amdhsa_code_object_version")
7025 return ParseDirectiveAMDHSACodeObjectVersion();
7029 return ParseDirectiveHSAMetadata();
7031 if (IDVal ==
".amd_kernel_code_t")
7032 return ParseDirectiveAMDKernelCodeT();
7034 if (IDVal ==
".amdgpu_hsa_kernel")
7035 return ParseDirectiveAMDGPUHsaKernel();
7037 if (IDVal ==
".amd_amdgpu_isa")
7038 return ParseDirectiveISAVersion();
7042 Twine(
" directive is "
7043 "not available on non-amdhsa OSes"))
7048 if (IDVal ==
".amdgcn_target")
7049 return ParseDirectiveAMDGCNTarget();
7051 if (IDVal ==
".amdgpu_lds")
7052 return ParseDirectiveAMDGPULDS();
7054 if (IDVal ==
".amdgpu_info")
7055 return ParseDirectiveAMDGPUInfo();
7058 return ParseDirectivePALMetadataBegin();
7061 return ParseDirectivePALMetadata();
7066bool AMDGPUAsmParser::subtargetHasRegister(
const MCRegisterInfo &MRI,
7073 return hasSGPR104_SGPR105();
7076 case SRC_SHARED_BASE_LO:
7077 case SRC_SHARED_BASE:
7078 case SRC_SHARED_LIMIT_LO:
7079 case SRC_SHARED_LIMIT:
7081 case SRC_PRIVATE_BASE_LO:
7082 case SRC_PRIVATE_BASE:
7083 case SRC_PRIVATE_LIMIT_LO:
7084 case SRC_PRIVATE_LIMIT:
7086 case SRC_FLAT_SCRATCH_BASE_LO:
7087 case SRC_FLAT_SCRATCH_BASE_HI:
7088 return hasGloballyAddressableScratch();
7089 case SRC_POPS_EXITING_WAVE_ID:
7102 getTargetStreamer().getTargetID()->isXnackSupported();
7132 return hasSGPR102_SGPR103();
7140 ParseStatus Res = parseVOPD(
Operands);
7145 Res = MatchOperandParserImpl(
Operands, Mnemonic);
7157 SMLoc LBraceLoc = getLoc();
7162 auto Loc = getLoc();
7165 Error(Loc,
"expected a register");
7169 RBraceLoc = getLoc();
7174 "expected a comma or a closing square bracket"))
7178 if (
Operands.size() - Prefix > 1) {
7180 AMDGPUOperand::CreateToken(
this,
"[", LBraceLoc));
7181 Operands.push_back(AMDGPUOperand::CreateToken(
this,
"]", RBraceLoc));
7190StringRef AMDGPUAsmParser::parseMnemonicSuffix(StringRef Name) {
7192 setForcedEncodingSize(0);
7193 setForcedDPP(
false);
7194 setForcedSDWA(
false);
7196 if (
Name.consume_back(
"_e64_dpp")) {
7198 setForcedEncodingSize(64);
7201 if (
Name.consume_back(
"_e64")) {
7202 setForcedEncodingSize(64);
7205 if (
Name.consume_back(
"_e32")) {
7206 setForcedEncodingSize(32);
7209 if (
Name.consume_back(
"_dpp")) {
7213 if (
Name.consume_back(
"_sdwa")) {
7214 setForcedSDWA(
true);
7222 unsigned VariantID);
7228 Name = parseMnemonicSuffix(Name);
7234 Operands.push_back(AMDGPUOperand::CreateToken(
this, Name, NameLoc));
7236 bool IsMIMG = Name.starts_with(
"image_");
7239 OperandMode
Mode = OperandMode_Default;
7241 Mode = OperandMode_NSA;
7245 checkUnsupportedInstruction(Name, NameLoc);
7246 if (!Parser.hasPendingError()) {
7249 :
"not a valid operand.";
7269ParseStatus AMDGPUAsmParser::parseTokenOp(StringRef Name,
7272 if (!trySkipId(Name))
7275 Operands.push_back(AMDGPUOperand::CreateToken(
this, Name, S));
7279ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
const char *Prefix,
7288ParseStatus AMDGPUAsmParser::parseIntWithPrefix(
7290 std::function<
bool(int64_t &)> ConvertResult) {
7294 ParseStatus Res = parseIntWithPrefix(Prefix,
Value);
7298 if (ConvertResult && !ConvertResult(
Value)) {
7299 Error(S,
"invalid " + StringRef(Prefix) +
" value.");
7302 Operands.push_back(AMDGPUOperand::CreateImm(
this,
Value, S, ImmTy));
7306ParseStatus AMDGPUAsmParser::parseOperandArrayWithPrefix(
7308 bool (*ConvertResult)(int64_t &)) {
7317 const unsigned MaxSize = 4;
7321 for (
int I = 0;; ++
I) {
7323 SMLoc Loc = getLoc();
7327 if (
Op != 0 &&
Op != 1)
7328 return Error(Loc,
"invalid " + StringRef(Prefix) +
" value.");
7335 if (
I + 1 == MaxSize)
7336 return Error(getLoc(),
"expected a closing square bracket");
7342 Operands.push_back(AMDGPUOperand::CreateImm(
this, Val, S, ImmTy));
7346ParseStatus AMDGPUAsmParser::parseNamedBit(StringRef Name,
7348 AMDGPUOperand::ImmTy ImmTy,
7349 bool IgnoreNegative) {
7353 if (trySkipId(Name)) {
7355 }
else if (trySkipId(
"no", Name)) {
7364 return Error(S,
"r128 modifier is not supported on this GPU");
7365 if (Name ==
"a16" && !
hasA16())
7366 return Error(S,
"a16 modifier is not supported on this GPU");
7368 if (Bit == 0 && Name ==
"gds") {
7371 return Error(S,
"nogds is not allowed");
7374 if (
isGFX9() && ImmTy == AMDGPUOperand::ImmTyA16)
7375 ImmTy = AMDGPUOperand::ImmTyR128A16;
7377 Operands.push_back(AMDGPUOperand::CreateImm(
this, Bit, S, ImmTy));
7381unsigned AMDGPUAsmParser::getCPolKind(StringRef Id, StringRef Mnemo,
7382 bool &Disabling)
const {
7383 Disabling =
Id.consume_front(
"no");
7386 return StringSwitch<unsigned>(Id)
7393 return StringSwitch<unsigned>(Id)
7403 SMLoc StringLoc = getLoc();
7405 int64_t CPolVal = 0;
7425 ResScope = parseScope(
Operands, Scope);
7438 if (trySkipId(
"nv")) {
7442 }
else if (trySkipId(
"no",
"nv")) {
7449 if (trySkipId(
"scale_offset")) {
7453 }
else if (trySkipId(
"no",
"scale_offset")) {
7466 Operands.push_back(AMDGPUOperand::CreateImm(
this, CPolVal, StringLoc,
7467 AMDGPUOperand::ImmTyCPol));
7472 SMLoc OpLoc = getLoc();
7473 unsigned Enabled = 0, Seen = 0;
7477 unsigned CPol = getCPolKind(getId(), Mnemo, Disabling);
7484 return Error(S,
"dlc modifier is not supported on this GPU");
7487 return Error(S,
"scc modifier is not supported on this GPU");
7490 return Error(S,
"duplicate cache policy modifier");
7502 AMDGPUOperand::CreateImm(
this,
Enabled, OpLoc, AMDGPUOperand::ImmTyCPol));
7511 ParseStatus Res = parseStringOrIntWithPrefix(
7512 Operands,
"scope", {
"SCOPE_CU",
"SCOPE_SE",
"SCOPE_DEV",
"SCOPE_SYS"},
7526 ParseStatus Res = parseStringWithPrefix(
"th",
Value, StringLoc);
7530 if (
Value ==
"TH_DEFAULT")
7532 else if (
Value ==
"TH_STORE_LU" ||
Value ==
"TH_LOAD_WB" ||
7533 Value ==
"TH_LOAD_NT_WB") {
7534 return Error(StringLoc,
"invalid th value");
7535 }
else if (
Value.consume_front(
"TH_ATOMIC_")) {
7537 }
else if (
Value.consume_front(
"TH_LOAD_")) {
7539 }
else if (
Value.consume_front(
"TH_STORE_")) {
7542 return Error(StringLoc,
"invalid th value");
7545 if (
Value ==
"BYPASS")
7550 TH |= StringSwitch<int64_t>(
Value)
7560 .Default(0xffffffff);
7562 TH |= StringSwitch<int64_t>(
Value)
7573 .Default(0xffffffff);
7576 if (TH == 0xffffffff)
7577 return Error(StringLoc,
"invalid th value");
7584 AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx,
7585 AMDGPUOperand::ImmTy ImmT, int64_t
Default = 0,
7586 std::optional<unsigned> InsertAt = std::nullopt) {
7587 auto i = OptionalIdx.find(ImmT);
7588 if (i != OptionalIdx.end()) {
7589 unsigned Idx = i->second;
7590 const AMDGPUOperand &
Op =
7591 static_cast<const AMDGPUOperand &
>(*
Operands[
Idx]);
7595 Op.addImmOperands(Inst, 1);
7597 if (InsertAt.has_value())
7604ParseStatus AMDGPUAsmParser::parseStringWithPrefix(StringRef Prefix,
7610 StringLoc = getLoc();
7615ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7621 SMLoc StringLoc = getLoc();
7625 Value = getTokenStr();
7629 if (
Value == Ids[IntVal])
7634 if (IntVal < 0 || IntVal >= (int64_t)Ids.
size())
7635 return Error(StringLoc,
"invalid " + Twine(Name) +
" value");
7640ParseStatus AMDGPUAsmParser::parseStringOrIntWithPrefix(
7642 AMDGPUOperand::ImmTy
Type) {
7646 ParseStatus Res = parseStringOrIntWithPrefix(
Operands, Name, Ids, IntVal);
7648 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S,
Type));
7657bool AMDGPUAsmParser::tryParseFmt(
const char *Pref, int64_t MaxVal,
7660 SMLoc Loc = getLoc();
7662 auto Res = parseIntWithPrefix(Pref, Val);
7668 if (Val < 0 || Val > MaxVal) {
7669 Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7678 AMDGPUOperand::ImmTy ImmTy) {
7679 const char *Pref =
"index_key";
7681 SMLoc Loc = getLoc();
7682 auto Res = parseIntWithPrefix(Pref, ImmVal);
7686 if ((ImmTy == AMDGPUOperand::ImmTyIndexKey16bit ||
7687 ImmTy == AMDGPUOperand::ImmTyIndexKey32bit) &&
7688 (ImmVal < 0 || ImmVal > 1))
7689 return Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7691 if (ImmTy == AMDGPUOperand::ImmTyIndexKey8bit && (ImmVal < 0 || ImmVal > 3))
7692 return Error(Loc, Twine(
"out of range ", StringRef(Pref)));
7694 Operands.push_back(AMDGPUOperand::CreateImm(
this, ImmVal, Loc, ImmTy));
7699 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey8bit);
7703 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey16bit);
7707 return tryParseIndexKey(
Operands, AMDGPUOperand::ImmTyIndexKey32bit);
7712 AMDGPUOperand::ImmTy
Type) {
7718 return tryParseMatrixFMT(
Operands,
"matrix_a_fmt",
7719 AMDGPUOperand::ImmTyMatrixAFMT);
7723 return tryParseMatrixFMT(
Operands,
"matrix_b_fmt",
7724 AMDGPUOperand::ImmTyMatrixBFMT);
7729 AMDGPUOperand::ImmTy
Type) {
7735 return tryParseMatrixScale(
Operands,
"matrix_a_scale",
7736 AMDGPUOperand::ImmTyMatrixAScale);
7740 return tryParseMatrixScale(
Operands,
"matrix_b_scale",
7741 AMDGPUOperand::ImmTyMatrixBScale);
7746 AMDGPUOperand::ImmTy
Type) {
7752 return tryParseMatrixScaleFmt(
Operands,
"matrix_a_scale_fmt",
7753 AMDGPUOperand::ImmTyMatrixAScaleFmt);
7757 return tryParseMatrixScaleFmt(
Operands,
"matrix_b_scale_fmt",
7758 AMDGPUOperand::ImmTyMatrixBScaleFmt);
7763ParseStatus AMDGPUAsmParser::parseDfmtNfmt(int64_t &
Format) {
7764 using namespace llvm::AMDGPU::MTBUFFormat;
7770 for (
int I = 0;
I < 2; ++
I) {
7771 if (Dfmt == DFMT_UNDEF && !tryParseFmt(
"dfmt", DFMT_MAX, Dfmt))
7774 if (Nfmt == NFMT_UNDEF && !tryParseFmt(
"nfmt", NFMT_MAX, Nfmt))
7779 if ((Dfmt == DFMT_UNDEF) != (Nfmt == NFMT_UNDEF) &&
7785 if (Dfmt == DFMT_UNDEF && Nfmt == NFMT_UNDEF)
7788 Dfmt = (Dfmt ==
DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
7789 Nfmt = (Nfmt ==
NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
7795ParseStatus AMDGPUAsmParser::parseUfmt(int64_t &
Format) {
7796 using namespace llvm::AMDGPU::MTBUFFormat;
7800 if (!tryParseFmt(
"format", UFMT_MAX, Fmt))
7803 if (Fmt == UFMT_UNDEF)
7810bool AMDGPUAsmParser::matchDfmtNfmt(int64_t &Dfmt, int64_t &Nfmt,
7811 StringRef FormatStr, SMLoc Loc) {
7812 using namespace llvm::AMDGPU::MTBUFFormat;
7816 if (
Format != DFMT_UNDEF) {
7822 if (
Format != NFMT_UNDEF) {
7827 Error(Loc,
"unsupported format");
7831ParseStatus AMDGPUAsmParser::parseSymbolicSplitFormat(StringRef FormatStr,
7834 using namespace llvm::AMDGPU::MTBUFFormat;
7838 if (!matchDfmtNfmt(Dfmt, Nfmt, FormatStr, FormatLoc))
7843 SMLoc Loc = getLoc();
7844 if (!parseId(Str,
"expected a format string") ||
7845 !matchDfmtNfmt(Dfmt, Nfmt, Str, Loc))
7847 if (Dfmt == DFMT_UNDEF)
7848 return Error(Loc,
"duplicate numeric format");
7849 if (Nfmt == NFMT_UNDEF)
7850 return Error(Loc,
"duplicate data format");
7853 Dfmt = (Dfmt ==
DFMT_UNDEF) ? DFMT_DEFAULT : Dfmt;
7854 Nfmt = (Nfmt ==
NFMT_UNDEF) ? NFMT_DEFAULT : Nfmt;
7858 if (Ufmt == UFMT_UNDEF)
7859 return Error(FormatLoc,
"unsupported format");
7868ParseStatus AMDGPUAsmParser::parseSymbolicUnifiedFormat(StringRef FormatStr,
7871 using namespace llvm::AMDGPU::MTBUFFormat;
7874 if (Id == UFMT_UNDEF)
7878 return Error(Loc,
"unified format is not supported on this GPU");
7884ParseStatus AMDGPUAsmParser::parseNumericFormat(int64_t &
Format) {
7885 using namespace llvm::AMDGPU::MTBUFFormat;
7886 SMLoc Loc = getLoc();
7891 return Error(Loc,
"out of range format");
7896ParseStatus AMDGPUAsmParser::parseSymbolicOrNumericFormat(int64_t &
Format) {
7897 using namespace llvm::AMDGPU::MTBUFFormat;
7903 StringRef FormatStr;
7904 SMLoc Loc = getLoc();
7905 if (!parseId(FormatStr,
"expected a format string"))
7908 auto Res = parseSymbolicUnifiedFormat(FormatStr, Loc,
Format);
7910 Res = parseSymbolicSplitFormat(FormatStr, Loc,
Format);
7920 return parseNumericFormat(
Format);
7924 using namespace llvm::AMDGPU::MTBUFFormat;
7928 SMLoc Loc = getLoc();
7938 AMDGPUOperand::CreateImm(
this,
Format, Loc, AMDGPUOperand::ImmTyFORMAT));
7957 Res = parseSymbolicOrNumericFormat(
Format);
7962 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands[
Size - 2]);
7963 assert(
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyFORMAT);
7970 return Error(getLoc(),
"duplicate format");
7976 parseIntWithPrefix(
"offset",
Operands, AMDGPUOperand::ImmTyOffset);
7978 Res = parseIntWithPrefix(
"inst_offset",
Operands,
7979 AMDGPUOperand::ImmTyInstOffset);
7986 parseNamedBit(
"r128",
Operands, AMDGPUOperand::ImmTyR128A16);
7988 Res = parseNamedBit(
"a16",
Operands, AMDGPUOperand::ImmTyA16);
7994 parseIntWithPrefix(
"blgp",
Operands, AMDGPUOperand::ImmTyBLGP);
7997 parseOperandArrayWithPrefix(
"neg",
Operands, AMDGPUOperand::ImmTyBLGP);
8007 OptionalImmIndexMap OptionalIdx;
8009 unsigned OperandIdx[4];
8010 unsigned EnMask = 0;
8013 for (
unsigned i = 1, e =
Operands.size(); i != e; ++i) {
8014 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
8019 OperandIdx[SrcIdx] = Inst.
size();
8020 Op.addRegOperands(Inst, 1);
8027 OperandIdx[SrcIdx] = Inst.
size();
8033 if (
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyExpTgt) {
8034 Op.addImmOperands(Inst, 1);
8038 if (
Op.isToken() && (
Op.getToken() ==
"done" ||
Op.getToken() ==
"row_en"))
8042 OptionalIdx[
Op.getImmTy()] = i;
8048 if (OptionalIdx.find(AMDGPUOperand::ImmTyExpCompr) != OptionalIdx.end()) {
8055 for (
auto i = 0; i < SrcIdx; ++i) {
8057 EnMask |= Compr ? (0x3 << i * 2) : (0x1 << i);
8063 AMDGPUOperand::ImmTyExpCompr);
8073 int64_t CntVal,
bool Saturate,
8079 IntVal =
encode(ISA, IntVal, CntVal);
8080 if (CntVal !=
decode(ISA, IntVal)) {
8082 IntVal =
encode(ISA, IntVal, -1);
8090bool AMDGPUAsmParser::parseCnt(int64_t &IntVal) {
8092 SMLoc CntLoc = getLoc();
8093 StringRef CntName = getTokenStr();
8100 SMLoc ValLoc = getLoc();
8107 if (CntName ==
"vmcnt" || CntName ==
"vmcnt_sat") {
8109 }
else if (CntName ==
"expcnt" || CntName ==
"expcnt_sat") {
8111 }
else if (CntName ==
"lgkmcnt" || CntName ==
"lgkmcnt_sat") {
8114 Error(CntLoc,
"invalid counter name " + CntName);
8119 Error(ValLoc,
"too large value for " + CntName);
8128 Error(getLoc(),
"expected a counter name");
8142 if (!parseCnt(Waitcnt))
8150 Operands.push_back(AMDGPUOperand::CreateImm(
this, Waitcnt, S));
8154bool AMDGPUAsmParser::parseDelay(int64_t &Delay) {
8155 SMLoc FieldLoc = getLoc();
8156 StringRef FieldName = getTokenStr();
8161 SMLoc ValueLoc = getLoc();
8168 if (FieldName ==
"instid0") {
8170 }
else if (FieldName ==
"instskip") {
8172 }
else if (FieldName ==
"instid1") {
8175 Error(FieldLoc,
"invalid field name " + FieldName);
8194 .Case(
"VALU_DEP_1", 1)
8195 .Case(
"VALU_DEP_2", 2)
8196 .Case(
"VALU_DEP_3", 3)
8197 .Case(
"VALU_DEP_4", 4)
8198 .Case(
"TRANS32_DEP_1", 5)
8199 .Case(
"TRANS32_DEP_2", 6)
8200 .Case(
"TRANS32_DEP_3", 7)
8201 .Case(
"FMA_ACCUM_CYCLE_1", 8)
8202 .Case(
"SALU_CYCLE_1", 9)
8203 .Case(
"SALU_CYCLE_2", 10)
8204 .Case(
"SALU_CYCLE_3", 11)
8212 Delay |=
Value << Shift;
8222 if (!parseDelay(Delay))
8230 Operands.push_back(AMDGPUOperand::CreateImm(
this, Delay, S));
8234bool AMDGPUOperand::isSWaitCnt()
const {
return isImm(); }
8236bool AMDGPUOperand::isSDelayALU()
const {
return isImm(); }
8242void AMDGPUAsmParser::depCtrError(SMLoc Loc,
int ErrorId,
8243 StringRef DepCtrName) {
8246 Error(Loc, Twine(
"invalid counter name ", DepCtrName));
8249 Error(Loc, Twine(DepCtrName,
" is not supported on this GPU"));
8252 Error(Loc, Twine(
"duplicate counter name ", DepCtrName));
8255 Error(Loc, Twine(
"invalid value for ", DepCtrName));
8262bool AMDGPUAsmParser::parseDepCtr(int64_t &DepCtr,
unsigned &UsedOprMask) {
8264 using namespace llvm::AMDGPU::DepCtr;
8266 SMLoc DepCtrLoc = getLoc();
8267 StringRef DepCtrName = getTokenStr();
8277 unsigned PrevOprMask = UsedOprMask;
8278 int CntVal =
encodeDepCtr(DepCtrName, ExprVal, UsedOprMask, getSTI());
8281 depCtrError(DepCtrLoc, CntVal, DepCtrName);
8290 Error(getLoc(),
"expected a counter name");
8295 int64_t CntValMask = PrevOprMask ^ UsedOprMask;
8296 DepCtr = (DepCtr & ~CntValMask) | CntVal;
8301 using namespace llvm::AMDGPU::DepCtr;
8304 SMLoc Loc = getLoc();
8307 unsigned UsedOprMask = 0;
8309 if (!parseDepCtr(DepCtr, UsedOprMask))
8317 Operands.push_back(AMDGPUOperand::CreateImm(
this, DepCtr, Loc));
8321bool AMDGPUOperand::isDepCtr()
const {
return isS16Imm(); }
8327ParseStatus AMDGPUAsmParser::parseHwregFunc(OperandInfoTy &HwReg,
8329 OperandInfoTy &Width) {
8330 using namespace llvm::AMDGPU::Hwreg;
8336 HwReg.Loc = getLoc();
8339 HwReg.IsSymbolic =
true;
8341 }
else if (!
parseExpr(HwReg.Val,
"a register name")) {
8349 if (!skipToken(
AsmToken::Comma,
"expected a comma or a closing parenthesis"))
8359 Width.Loc = getLoc();
8368 using namespace llvm::AMDGPU::Hwreg;
8371 SMLoc Loc = getLoc();
8373 StructuredOpField HwReg(
"id",
"hardware register", HwregId::Width,
8375 StructuredOpField
Offset(
"offset",
"bit offset", HwregOffset::Width,
8376 HwregOffset::Default);
8377 struct : StructuredOpField {
8378 using StructuredOpField::StructuredOpField;
8379 bool validate(AMDGPUAsmParser &Parser)
const override {
8381 return Error(Parser,
"only values from 1 to 32 are legal");
8384 } Width(
"size",
"bitfield width", HwregSize::Width, HwregSize::Default);
8385 ParseStatus Res = parseStructuredOpFields({&HwReg, &
Offset, &Width});
8388 Res = parseHwregFunc(HwReg,
Offset, Width);
8391 if (!validateStructuredOpFields({&HwReg, &
Offset, &Width}))
8393 ImmVal = HwregEncoding::encode(HwReg.Val,
Offset.Val, Width.Val);
8397 parseExpr(ImmVal,
"a hwreg macro, structured immediate"))
8404 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8406 AMDGPUOperand::CreateImm(
this, ImmVal, Loc, AMDGPUOperand::ImmTyHwreg));
8410bool AMDGPUOperand::isHwreg()
const {
return isImmTy(ImmTyHwreg); }
8416bool AMDGPUAsmParser::parseSendMsgBody(OperandInfoTy &
Msg, OperandInfoTy &
Op,
8417 OperandInfoTy &Stream) {
8418 using namespace llvm::AMDGPU::SendMsg;
8423 Msg.IsSymbolic =
true;
8430 Op.IsDefined =
true;
8436 }
else if (!
parseExpr(
Op.Val,
"an operation name")) {
8441 Stream.IsDefined =
true;
8442 Stream.Loc = getLoc();
8451bool AMDGPUAsmParser::validateSendMsg(
const OperandInfoTy &
Msg,
8452 const OperandInfoTy &
Op,
8453 const OperandInfoTy &Stream) {
8454 using namespace llvm::AMDGPU::SendMsg;
8463 Error(
Msg.Loc,
"specified message id is not supported on this GPU");
8468 Error(
Msg.Loc,
"invalid message id");
8474 Error(
Op.Loc,
"message does not support operations");
8476 Error(
Msg.Loc,
"missing message operation");
8482 Error(
Op.Loc,
"specified operation id is not supported on this GPU");
8484 Error(
Op.Loc,
"invalid operation id");
8489 Error(Stream.Loc,
"message operation does not support streams");
8493 Error(Stream.Loc,
"invalid message stream id");
8500 using namespace llvm::AMDGPU::SendMsg;
8503 SMLoc Loc = getLoc();
8507 OperandInfoTy
Op(OP_NONE_);
8508 OperandInfoTy Stream(STREAM_ID_NONE_);
8509 if (parseSendMsgBody(
Msg,
Op, Stream) && validateSendMsg(
Msg,
Op, Stream)) {
8514 }
else if (
parseExpr(ImmVal,
"a sendmsg macro")) {
8516 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8522 AMDGPUOperand::CreateImm(
this, ImmVal, Loc, AMDGPUOperand::ImmTySendMsg));
8526bool AMDGPUOperand::isSendMsg()
const {
return isImmTy(ImmTySendMsg); }
8529 using namespace llvm::AMDGPU::WaitEvent;
8531 SMLoc Loc = getLoc();
8534 StructuredOpField DontWaitExportReady(
"dont_wait_export_ready",
"bit value",
8536 StructuredOpField ExportReady(
"export_ready",
"bit value", 1, 0);
8538 StructuredOpField *TargetBitfield =
8539 isGFX11() ? &DontWaitExportReady : &ExportReady;
8541 ParseStatus Res = parseStructuredOpFields({TargetBitfield});
8545 if (!validateStructuredOpFields({TargetBitfield}))
8547 ImmVal = TargetBitfield->Val;
8554 return Error(Loc,
"invalid immediate: only 16-bit values are legal");
8556 Operands.push_back(AMDGPUOperand::CreateImm(
this, ImmVal, Loc,
8557 AMDGPUOperand::ImmTyWaitEvent));
8561bool AMDGPUOperand::isWaitEvent()
const {
return isImmTy(ImmTyWaitEvent); }
8574 int Slot = StringSwitch<int>(Str)
8581 return Error(S,
"invalid interpolation slot");
8584 AMDGPUOperand::CreateImm(
this, Slot, S, AMDGPUOperand::ImmTyInterpSlot));
8595 if (!Str.starts_with(
"attr"))
8596 return Error(S,
"invalid interpolation attribute");
8598 StringRef Chan = Str.take_back(2);
8599 int AttrChan = StringSwitch<int>(Chan)
8606 return Error(S,
"invalid or missing interpolation attribute channel");
8608 Str = Str.drop_back(2).drop_front(4);
8611 if (Str.getAsInteger(10, Attr))
8612 return Error(S,
"invalid or missing interpolation attribute number");
8615 return Error(S,
"out of bounds interpolation attribute number");
8620 AMDGPUOperand::CreateImm(
this, Attr, S, AMDGPUOperand::ImmTyInterpAttr));
8621 Operands.push_back(AMDGPUOperand::CreateImm(
8622 this, AttrChan, SChan, AMDGPUOperand::ImmTyInterpAttrChan));
8631 using namespace llvm::AMDGPU::Exp;
8641 return Error(S, (Id == ET_INVALID)
8642 ?
"invalid exp target"
8643 :
"exp target is not supported on this GPU");
8646 AMDGPUOperand::CreateImm(
this, Id, S, AMDGPUOperand::ImmTyExpTgt));
8654bool AMDGPUAsmParser::isId(
const AsmToken &Token,
const StringRef Id)
const {
8658bool AMDGPUAsmParser::isId(
const StringRef Id)
const {
8663 return getTokenKind() ==
Kind;
8666StringRef AMDGPUAsmParser::getId()
const {
8670bool AMDGPUAsmParser::trySkipId(
const StringRef Id) {
8678bool AMDGPUAsmParser::trySkipId(
const StringRef Pref,
const StringRef Id) {
8680 StringRef Tok = getTokenStr();
8689bool AMDGPUAsmParser::trySkipId(
const StringRef Id,
8691 if (isId(Id) && peekToken().is(Kind)) {
8700 if (isToken(Kind)) {
8708 const StringRef ErrMsg) {
8709 if (!trySkipToken(Kind)) {
8710 Error(getLoc(), ErrMsg);
8716bool AMDGPUAsmParser::parseExpr(int64_t &
Imm, StringRef Expected) {
8720 if (Parser.parseExpression(Expr))
8723 if (Expr->evaluateAsAbsolute(
Imm))
8726 if (Expected.empty()) {
8727 Error(S,
"expected absolute expression");
8730 Twine(
"expected ", Expected) + Twine(
" or an absolute expression"));
8739 if (Parser.parseExpression(Expr))
8743 if (Expr->evaluateAsAbsolute(IntVal)) {
8744 Operands.push_back(AMDGPUOperand::CreateImm(
this, IntVal, S));
8746 Operands.push_back(AMDGPUOperand::CreateExpr(
this, Expr, S));
8751bool AMDGPUAsmParser::parseString(StringRef &Val,
const StringRef ErrMsg) {
8753 Val =
getToken().getStringContents();
8757 Error(getLoc(), ErrMsg);
8761bool AMDGPUAsmParser::parseId(StringRef &Val,
const StringRef ErrMsg) {
8763 Val = getTokenStr();
8767 if (!ErrMsg.
empty())
8768 Error(getLoc(), ErrMsg);
8772AsmToken AMDGPUAsmParser::getToken()
const {
return Parser.getTok(); }
8774AsmToken AMDGPUAsmParser::peekToken(
bool ShouldSkipSpace) {
8777 : getLexer().peekTok(ShouldSkipSpace);
8781 auto TokCount = getLexer().peekTokens(Tokens);
8788 return getLexer().getKind();
8791SMLoc AMDGPUAsmParser::getLoc()
const {
return getToken().getLoc(); }
8793StringRef AMDGPUAsmParser::getTokenStr()
const {
8797void AMDGPUAsmParser::lex() { Parser.Lex(); }
8799const AMDGPUOperand &
8801 int MCOpIdx)
const {
8803 const AMDGPUOperand &TargetOp =
static_cast<AMDGPUOperand &
>(*Op);
8804 if (TargetOp.getMCOpIdx() == MCOpIdx)
8811 return ((AMDGPUOperand &)*
Operands[0]).getStartLoc();
8815SMLoc AMDGPUAsmParser::getLaterLoc(SMLoc a, SMLoc b) {
8820 int MCOpIdx)
const {
8821 return findMCOperand(
Operands, MCOpIdx).getStartLoc();
8824SMLoc AMDGPUAsmParser::getOperandLoc(
8825 std::function<
bool(
const AMDGPUOperand &)>
Test,
8827 for (
unsigned i =
Operands.size() - 1; i > 0; --i) {
8828 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
8830 return Op.getStartLoc();
8835SMLoc AMDGPUAsmParser::getImmLoc(AMDGPUOperand::ImmTy
Type,
8837 auto Test = [=](
const AMDGPUOperand &
Op) {
return Op.isImmTy(
Type); };
8852 StringRef
Id = getTokenStr();
8853 SMLoc IdLoc = getLoc();
8859 find_if(Fields, [Id](StructuredOpField *
F) {
return F->Id ==
Id; });
8860 if (
I == Fields.
end())
8861 return Error(IdLoc,
"unknown field");
8862 if ((*I)->IsDefined)
8863 return Error(IdLoc,
"duplicate field");
8866 (*I)->Loc = getLoc();
8869 (*I)->IsDefined =
true;
8876bool AMDGPUAsmParser::validateStructuredOpFields(
8878 return all_of(Fields, [
this](
const StructuredOpField *
F) {
8879 return F->validate(*
this);
8889 const unsigned XorMask) {
8896bool AMDGPUAsmParser::parseSwizzleOperand(int64_t &
Op,
const unsigned MinVal,
8897 const unsigned MaxVal,
8898 const Twine &ErrMsg, SMLoc &Loc) {
8914bool AMDGPUAsmParser::parseSwizzleOperands(
const unsigned OpNum, int64_t *
Op,
8915 const unsigned MinVal,
8916 const unsigned MaxVal,
8917 const StringRef ErrMsg) {
8919 for (
unsigned i = 0; i < OpNum; ++i) {
8920 if (!parseSwizzleOperand(
Op[i], MinVal, MaxVal, ErrMsg, Loc))
8927bool AMDGPUAsmParser::parseSwizzleQuadPerm(int64_t &
Imm) {
8928 using namespace llvm::AMDGPU::Swizzle;
8931 if (parseSwizzleOperands(LANE_NUM, Lane, 0, LANE_MAX,
8932 "expected a 2-bit lane id")) {
8942bool AMDGPUAsmParser::parseSwizzleBroadcast(int64_t &
Imm) {
8943 using namespace llvm::AMDGPU::Swizzle;
8949 if (!parseSwizzleOperand(GroupSize, 2, 32,
8950 "group size must be in the interval [2,32]", Loc)) {
8954 Error(Loc,
"group size must be a power of two");
8957 if (parseSwizzleOperand(LaneIdx, 0, GroupSize - 1,
8958 "lane id must be in the interval [0,group size - 1]",
8966bool AMDGPUAsmParser::parseSwizzleReverse(int64_t &
Imm) {
8967 using namespace llvm::AMDGPU::Swizzle;
8972 if (!parseSwizzleOperand(GroupSize, 2, 32,
8973 "group size must be in the interval [2,32]", Loc)) {
8977 Error(Loc,
"group size must be a power of two");
8985bool AMDGPUAsmParser::parseSwizzleSwap(int64_t &
Imm) {
8986 using namespace llvm::AMDGPU::Swizzle;
8991 if (!parseSwizzleOperand(GroupSize, 1, 16,
8992 "group size must be in the interval [1,16]", Loc)) {
8996 Error(Loc,
"group size must be a power of two");
9004bool AMDGPUAsmParser::parseSwizzleBitmaskPerm(int64_t &
Imm) {
9005 using namespace llvm::AMDGPU::Swizzle;
9012 SMLoc StrLoc = getLoc();
9013 if (!parseString(Ctl)) {
9016 if (Ctl.
size() != BITMASK_WIDTH) {
9017 Error(StrLoc,
"expected a 5-character mask");
9021 unsigned AndMask = 0;
9022 unsigned OrMask = 0;
9023 unsigned XorMask = 0;
9025 for (
size_t i = 0; i < Ctl.
size(); ++i) {
9029 Error(StrLoc,
"invalid mask");
9050bool AMDGPUAsmParser::parseSwizzleFFT(int64_t &
Imm) {
9051 using namespace llvm::AMDGPU::Swizzle;
9054 Error(getLoc(),
"FFT mode swizzle not supported on this GPU");
9060 if (!parseSwizzleOperand(Swizzle, 0, FFT_SWIZZLE_MAX,
9061 "FFT swizzle must be in the interval [0," +
9062 Twine(FFT_SWIZZLE_MAX) + Twine(
']'),
9070bool AMDGPUAsmParser::parseSwizzleRotate(int64_t &
Imm) {
9071 using namespace llvm::AMDGPU::Swizzle;
9074 Error(getLoc(),
"Rotate mode swizzle not supported on this GPU");
9081 if (!parseSwizzleOperand(
Direction, 0, 1,
9082 "direction must be 0 (left) or 1 (right)", Loc))
9086 if (!parseSwizzleOperand(
9087 RotateSize, 0, ROTATE_MAX_SIZE,
9088 "number of threads to rotate must be in the interval [0," +
9089 Twine(ROTATE_MAX_SIZE) + Twine(
']'),
9094 (RotateSize << ROTATE_SIZE_SHIFT);
9098bool AMDGPUAsmParser::parseSwizzleOffset(int64_t &
Imm) {
9100 SMLoc OffsetLoc = getLoc();
9106 Error(OffsetLoc,
"expected a 16-bit offset");
9112bool AMDGPUAsmParser::parseSwizzleMacro(int64_t &
Imm) {
9113 using namespace llvm::AMDGPU::Swizzle;
9117 SMLoc ModeLoc = getLoc();
9120 if (trySkipId(IdSymbolic[ID_QUAD_PERM])) {
9121 Ok = parseSwizzleQuadPerm(
Imm);
9122 }
else if (trySkipId(IdSymbolic[ID_BITMASK_PERM])) {
9123 Ok = parseSwizzleBitmaskPerm(
Imm);
9124 }
else if (trySkipId(IdSymbolic[ID_BROADCAST])) {
9125 Ok = parseSwizzleBroadcast(
Imm);
9126 }
else if (trySkipId(IdSymbolic[ID_SWAP])) {
9127 Ok = parseSwizzleSwap(
Imm);
9128 }
else if (trySkipId(IdSymbolic[ID_REVERSE])) {
9129 Ok = parseSwizzleReverse(
Imm);
9130 }
else if (trySkipId(IdSymbolic[ID_FFT])) {
9131 Ok = parseSwizzleFFT(
Imm);
9132 }
else if (trySkipId(IdSymbolic[ID_ROTATE])) {
9133 Ok = parseSwizzleRotate(
Imm);
9135 Error(ModeLoc,
"expected a swizzle mode");
9138 return Ok && skipToken(
AsmToken::RParen,
"expected a closing parentheses");
9148 if (trySkipId(
"offset")) {
9152 if (trySkipId(
"swizzle")) {
9153 Ok = parseSwizzleMacro(
Imm);
9155 Ok = parseSwizzleOffset(
Imm);
9160 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTySwizzle));
9167bool AMDGPUOperand::isSwizzle()
const {
return isImmTy(ImmTySwizzle); }
9173int64_t AMDGPUAsmParser::parseGPRIdxMacro() {
9175 using namespace llvm::AMDGPU::VGPRIndexMode;
9187 for (
unsigned ModeId = ID_MIN; ModeId <=
ID_MAX; ++ModeId) {
9188 if (trySkipId(IdSymbolic[ModeId])) {
9196 ?
"expected a VGPR index mode or a closing parenthesis"
9197 :
"expected a VGPR index mode");
9202 Error(S,
"duplicate VGPR index mode");
9210 "expected a comma or a closing parenthesis"))
9219 using namespace llvm::AMDGPU::VGPRIndexMode;
9225 Imm = parseGPRIdxMacro();
9229 if (getParser().parseAbsoluteExpression(
Imm))
9232 return Error(S,
"invalid immediate: only 4-bit values are legal");
9236 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTyGprIdxMode));
9240bool AMDGPUOperand::isGPRIdxMode()
const {
return isImmTy(ImmTyGprIdxMode); }
9251 if (isRegister() || isModifier())
9258 assert(Opr.isImm() || Opr.isExpr());
9259 SMLoc Loc = Opr.getStartLoc();
9263 if (Opr.isExpr() && !Opr.isSymbolRefExpr()) {
9264 Error(Loc,
"expected an absolute expression or a label");
9265 }
else if (Opr.isImm() && !Opr.isS16Imm()) {
9266 Error(Loc,
"expected a 16-bit signed jump offset");
9286 OptionalImmIndexMap OptionalIdx;
9287 unsigned FirstOperandIdx = 1;
9288 bool IsAtomicReturn =
false;
9294 for (
unsigned i = FirstOperandIdx, e =
Operands.size(); i != e; ++i) {
9295 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
9299 Op.addRegOperands(Inst, 1);
9303 if (IsAtomicReturn && i == FirstOperandIdx)
9304 Op.addRegOperands(Inst, 1);
9309 if (
Op.isImm() &&
Op.getImmTy() == AMDGPUOperand::ImmTyNone) {
9310 Op.addImmOperands(Inst, 1);
9322 OptionalIdx[
Op.getImmTy()] = i;
9326 AMDGPUOperand::ImmTyOffset);
9342bool AMDGPUOperand::isSMRDOffset8()
const {
9346bool AMDGPUOperand::isSMEMOffset()
const {
9348 return isImmLiteral();
9351bool AMDGPUOperand::isSMRDLiteralOffset()
const {
9386bool AMDGPUAsmParser::convertDppBoundCtrl(int64_t &BoundCtrl) {
9387 if (BoundCtrl == 0 || BoundCtrl == 1) {
9395void AMDGPUAsmParser::onBeginOfFile() {
9396 if (!getParser().getStreamer().getTargetStreamer())
9399 if (!getTargetStreamer().getTargetID())
9400 getTargetStreamer().initializeTargetID(getSTI(),
9404void AMDGPUAsmParser::emitTargetDirective() {
9405 if (TargetDirectiveEmitted)
9407 TargetDirectiveEmitted =
true;
9409 if (!getParser().getStreamer().getTargetStreamer() ||
9414 getTargetStreamer().EmitDirectiveAMDGCNTarget();
9423bool AMDGPUAsmParser::parsePrimaryExpr(
const MCExpr *&Res, SMLoc &EndLoc) {
9427 StringRef TokenId = getTokenStr();
9428 AGVK VK = StringSwitch<AGVK>(TokenId)
9429 .Case(
"max", AGVK::AGVK_Max)
9430 .Case(
"min", AGVK::AGVK_Min)
9431 .Case(
"or", AGVK::AGVK_Or)
9432 .Case(
"extrasgprs", AGVK::AGVK_ExtraSGPRs)
9433 .Case(
"totalnumvgprs", AGVK::AGVK_TotalNumVGPRs)
9434 .Case(
"alignto", AGVK::AGVK_AlignTo)
9435 .Case(
"occupancy", AGVK::AGVK_Occupancy)
9436 .Case(
"instprefsize", AGVK::AGVK_InstPrefSize)
9437 .Default(AGVK::AGVK_None);
9446 if (Exprs.
empty()) {
9448 "empty " + Twine(TokenId) +
" expression");
9451 if (CommaCount + 1 != Exprs.
size()) {
9453 "mismatch of commas in " + Twine(TokenId) +
" expression");
9457 Expected && Exprs.
size() != Expected) {
9458 Error(
getToken().getLoc(), Twine(TokenId) +
" expression expects " +
9459 Twine(Expected) +
" operands");
9466 if (getParser().parseExpression(Expr, EndLoc))
9470 if (LastTokenWasComma)
9474 "unexpected token in " + Twine(TokenId) +
" expression");
9480 return getParser().parsePrimaryExpr(Res, EndLoc,
nullptr);
9484 StringRef
Name = getTokenStr();
9485 if (Name ==
"mul") {
9486 return parseIntWithPrefix(
"mul",
Operands, AMDGPUOperand::ImmTyOModSI,
9490 if (Name ==
"div") {
9491 return parseIntWithPrefix(
"div",
Operands, AMDGPUOperand::ImmTyOModSI,
9502 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9507 const AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9508 AMDGPU::OpName::src2};
9516 int DstIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst);
9521 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src0_modifiers);
9523 if (
DstOp.isReg() &&
9528 if ((OpSel & (1 << SrcNum)) != 0)
9534void AMDGPUAsmParser::cvtVOP3OpSel(MCInst &Inst,
9541 OptionalImmIndexMap &OptionalIdx) {
9542 cvtVOP3P(Inst,
Operands, OptionalIdx);
9551 &&
Desc.NumOperands > (OpNum + 1)
9553 &&
Desc.operands()[OpNum + 1].RegClass != -1
9555 &&
Desc.getOperandConstraint(OpNum + 1,
9559void AMDGPUAsmParser::cvtOpSelHelper(MCInst &Inst,
unsigned OpSel) {
9561 constexpr AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9562 AMDGPU::OpName::src2};
9563 constexpr AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9564 AMDGPU::OpName::src1_modifiers,
9565 AMDGPU::OpName::src2_modifiers};
9566 for (
int J = 0; J < 3; ++J) {
9567 int OpIdx = AMDGPU::getNamedOperandIdx(
Opc,
Ops[J]);
9573 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
9576 if ((OpSel & (1 << J)) != 0)
9579 if (ModOps[J] == AMDGPU::OpName::src0_modifiers && (OpSel & (1 << 3)) != 0)
9586void AMDGPUAsmParser::cvtVOP3Interp(MCInst &Inst,
9588 OptionalImmIndexMap OptionalIdx;
9593 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9594 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9598 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9600 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9601 }
else if (
Op.isInterpSlot() ||
Op.isInterpAttr() ||
9602 Op.isInterpAttrChan()) {
9604 }
else if (
Op.isImmModifier()) {
9605 OptionalIdx[
Op.getImmTy()] =
I;
9613 AMDGPUOperand::ImmTyHigh);
9617 AMDGPUOperand::ImmTyClamp);
9621 AMDGPUOperand::ImmTyOModSI);
9626 AMDGPUOperand::ImmTyOpSel);
9627 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9630 cvtOpSelHelper(Inst, OpSel);
9635 OptionalImmIndexMap OptionalIdx;
9640 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9641 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9645 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9647 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9648 }
else if (
Op.isImmModifier()) {
9649 OptionalIdx[
Op.getImmTy()] =
I;
9657 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9660 AMDGPUOperand::ImmTyOpSel);
9663 AMDGPUOperand::ImmTyWaitEXP);
9669 cvtOpSelHelper(Inst, OpSel);
9672void AMDGPUAsmParser::cvtScaledMFMA(MCInst &Inst,
9674 OptionalImmIndexMap OptionalIdx;
9677 int CbszOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::cbsz);
9681 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J)
9682 static_cast<AMDGPUOperand &
>(*
Operands[
I++]).addRegOperands(Inst, 1);
9685 AMDGPUOperand &
Op =
static_cast<AMDGPUOperand &
>(*
Operands[
I]);
9690 if (NumOperands == CbszOpIdx) {
9695 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9696 }
else if (
Op.isImmModifier()) {
9697 OptionalIdx[
Op.getImmTy()] =
I;
9699 Op.addRegOrImmOperands(Inst, 1);
9704 auto CbszIdx = OptionalIdx.find(AMDGPUOperand::ImmTyCBSZ);
9705 if (CbszIdx != OptionalIdx.end()) {
9706 int CbszVal = ((AMDGPUOperand &)*
Operands[CbszIdx->second]).
getImm();
9710 int BlgpOpIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::blgp);
9711 auto BlgpIdx = OptionalIdx.find(AMDGPUOperand::ImmTyBLGP);
9712 if (BlgpIdx != OptionalIdx.end()) {
9713 int BlgpVal = ((AMDGPUOperand &)*
Operands[BlgpIdx->second]).
getImm();
9724 auto OpselIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSel);
9725 if (OpselIdx != OptionalIdx.end()) {
9726 OpSel =
static_cast<const AMDGPUOperand &
>(*
Operands[OpselIdx->second])
9730 unsigned OpSelHi = 0;
9731 auto OpselHiIdx = OptionalIdx.find(AMDGPUOperand::ImmTyOpSelHi);
9732 if (OpselHiIdx != OptionalIdx.end()) {
9733 OpSelHi =
static_cast<const AMDGPUOperand &
>(*
Operands[OpselHiIdx->second])
9736 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9737 AMDGPU::OpName::src1_modifiers};
9739 for (
unsigned J = 0; J < 2; ++J) {
9740 unsigned ModVal = 0;
9741 if (OpSel & (1 << J))
9743 if (OpSelHi & (1 << J))
9746 const int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
9752 OptionalImmIndexMap &OptionalIdx) {
9757 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
9758 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
9762 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
9764 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
9765 }
else if (
Op.isImmModifier()) {
9766 OptionalIdx[
Op.getImmTy()] =
I;
9768 Op.addRegOrImmOperands(Inst, 1);
9774 AMDGPUOperand::ImmTyScaleSel);
9778 AMDGPUOperand::ImmTyClamp);
9784 AMDGPUOperand::ImmTyByteSel);
9789 AMDGPUOperand::ImmTyOModSI);
9796 auto *it = Inst.
begin();
9798 it, AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2_modifiers));
9807 OptionalImmIndexMap OptionalIdx;
9808 cvtVOP3(Inst,
Operands, OptionalIdx);
9812 OptionalImmIndexMap &OptIdx) {
9817 if (
Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_F16_vi ||
9818 Opc == AMDGPU::V_CVT_SCALEF32_PK_FP4_BF16_vi ||
9819 Opc == AMDGPU::V_CVT_SR_BF8_F32_vi ||
9820 Opc == AMDGPU::V_CVT_SR_FP8_F32_vi ||
9821 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx11 ||
9822 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx11 ||
9823 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx12 ||
9824 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx12 ||
9825 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_gfx13 ||
9826 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_gfx13) {
9835 int VdstInIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
9836 if (VdstInIdx != -1 && VdstInIdx ==
static_cast<int>(Inst.
getNumOperands()))
9839 int BitOp3Idx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::bitop3);
9840 if (BitOp3Idx != -1) {
9847 int OpSelIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel);
9848 if (OpSelIdx != -1) {
9852 int OpSelHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::op_sel_hi);
9853 if (OpSelHiIdx != -1) {
9854 int DefaultVal =
IsPacked ? -1 : 0;
9860 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_fmt);
9861 if (MatrixAFMTIdx != -1) {
9863 AMDGPUOperand::ImmTyMatrixAFMT, 0);
9867 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_fmt);
9868 if (MatrixBFMTIdx != -1) {
9870 AMDGPUOperand::ImmTyMatrixBFMT, 0);
9873 int MatrixAScaleIdx =
9874 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale);
9875 if (MatrixAScaleIdx != -1) {
9877 AMDGPUOperand::ImmTyMatrixAScale, 0);
9880 int MatrixBScaleIdx =
9881 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale);
9882 if (MatrixBScaleIdx != -1) {
9884 AMDGPUOperand::ImmTyMatrixBScale, 0);
9887 int MatrixAScaleFmtIdx =
9888 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_a_scale_fmt);
9889 if (MatrixAScaleFmtIdx != -1) {
9891 AMDGPUOperand::ImmTyMatrixAScaleFmt, 0);
9894 int MatrixBScaleFmtIdx =
9895 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::matrix_b_scale_fmt);
9896 if (MatrixBScaleFmtIdx != -1) {
9898 AMDGPUOperand::ImmTyMatrixBScaleFmt, 0);
9903 AMDGPUOperand::ImmTyMatrixAReuse, 0);
9907 AMDGPUOperand::ImmTyMatrixBReuse, 0);
9909 int NegLoIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::neg_lo);
9913 int NegHiIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::neg_hi);
9917 const AMDGPU::OpName
Ops[] = {AMDGPU::OpName::src0, AMDGPU::OpName::src1,
9918 AMDGPU::OpName::src2};
9919 const AMDGPU::OpName ModOps[] = {AMDGPU::OpName::src0_modifiers,
9920 AMDGPU::OpName::src1_modifiers,
9921 AMDGPU::OpName::src2_modifiers};
9924 unsigned OpSelHi = 0;
9931 if (OpSelHiIdx != -1)
9940 for (
int J = 0; J < 3; ++J) {
9941 int OpIdx = AMDGPU::getNamedOperandIdx(
Opc,
Ops[J]);
9945 int ModIdx = AMDGPU::getNamedOperandIdx(
Opc, ModOps[J]);
9955 uint32_t ModVal = 0;
9957 const MCOperand &SrcOp = Inst.
getOperand(OpIdx);
9958 if (SrcOp.
isReg() && getMRI()
9965 if ((OpSel & (1 << J)) != 0)
9969 if ((OpSelHi & (1 << J)) != 0)
9972 if ((NegLo & (1 << J)) != 0)
9975 if ((NegHi & (1 << J)) != 0)
9983 OptionalImmIndexMap OptIdx;
9989 unsigned i,
unsigned Opc,
9991 if (AMDGPU::getNamedOperandIdx(
Opc,
OpName) != -1)
9992 ((AMDGPUOperand &)*
Operands[i]).addRegOrImmWithFPInputModsOperands(Inst, 2);
9994 ((AMDGPUOperand &)*
Operands[i]).addRegOperands(Inst, 1);
10000 ((AMDGPUOperand &)*
Operands[1]).addRegOperands(Inst, 1);
10003 ((AMDGPUOperand &)*
Operands[1]).addRegOperands(Inst, 1);
10004 ((AMDGPUOperand &)*
Operands[4]).addRegOperands(Inst, 1);
10006 OptionalImmIndexMap OptIdx;
10007 for (
unsigned i = 5; i <
Operands.size(); ++i) {
10008 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[i]);
10009 OptIdx[
Op.getImmTy()] = i;
10014 AMDGPUOperand::ImmTyIndexKey8bit);
10018 AMDGPUOperand::ImmTyIndexKey16bit);
10022 AMDGPUOperand::ImmTyIndexKey32bit);
10039 SMLoc S = getLoc();
10042 Operands.push_back(AMDGPUOperand::CreateToken(
this,
"::", S));
10043 SMLoc OpYLoc = getLoc();
10046 Operands.push_back(AMDGPUOperand::CreateToken(
this, OpYName, OpYLoc));
10049 return Error(OpYLoc,
"expected a VOPDY instruction after ::");
10058 auto addOp = [&](uint16_t ParsedOprIdx) {
10059 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[ParsedOprIdx]);
10061 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10065 Op.addRegOperands(Inst, 1);
10068 if (
Op.isImm() ||
Op.isExpr()) {
10069 Op.addImmOperands(Inst, 1);
10081 addOp(InstInfo[CompIdx].getIndexOfDstInParsedOperands());
10085 const auto &CInfo = InstInfo[CompIdx];
10086 auto CompSrcOperandsNum = InstInfo[CompIdx].getCompParsedSrcOperandsNum();
10087 for (
unsigned CompSrcIdx = 0; CompSrcIdx < CompSrcOperandsNum; ++CompSrcIdx)
10088 addOp(CInfo.getIndexOfSrcInParsedOperands(CompSrcIdx));
10089 if (CInfo.hasSrc2Acc())
10090 addOp(CInfo.getIndexOfDstInParsedOperands());
10094 AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::bitop3);
10095 if (BitOp3Idx != -1) {
10096 OptionalImmIndexMap OptIdx;
10097 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands.back());
10099 OptIdx[
Op.getImmTy()] =
Operands.size() - 1;
10109bool AMDGPUOperand::isDPP8()
const {
return isImmTy(ImmTyDPP8); }
10111bool AMDGPUOperand::isDPPCtrl()
const {
10112 using namespace AMDGPU::DPP;
10114 bool result = isImm() && getImmTy() == ImmTyDppCtrl &&
isUInt<9>(
getImm());
10117 return (
Imm >= DppCtrl::QUAD_PERM_FIRST &&
10118 Imm <= DppCtrl::QUAD_PERM_LAST) ||
10119 (
Imm >= DppCtrl::ROW_SHL_FIRST &&
Imm <= DppCtrl::ROW_SHL_LAST) ||
10120 (
Imm >= DppCtrl::ROW_SHR_FIRST &&
Imm <= DppCtrl::ROW_SHR_LAST) ||
10121 (
Imm >= DppCtrl::ROW_ROR_FIRST &&
Imm <= DppCtrl::ROW_ROR_LAST) ||
10122 (
Imm == DppCtrl::WAVE_SHL1) || (
Imm == DppCtrl::WAVE_ROL1) ||
10123 (
Imm == DppCtrl::WAVE_SHR1) || (
Imm == DppCtrl::WAVE_ROR1) ||
10124 (
Imm == DppCtrl::ROW_MIRROR) || (
Imm == DppCtrl::ROW_HALF_MIRROR) ||
10125 (
Imm == DppCtrl::BCAST15) || (
Imm == DppCtrl::BCAST31) ||
10126 (
Imm >= DppCtrl::ROW_SHARE_FIRST &&
10127 Imm <= DppCtrl::ROW_SHARE_LAST) ||
10128 (
Imm >= DppCtrl::ROW_XMASK_FIRST &&
Imm <= DppCtrl::ROW_XMASK_LAST);
10137bool AMDGPUOperand::isBLGP()
const {
10141bool AMDGPUOperand::isS16Imm()
const {
10145bool AMDGPUOperand::isU16Imm()
const {
10153bool AMDGPUAsmParser::parseDimId(
unsigned &Encoding) {
10158 SMLoc Loc =
getToken().getEndLoc();
10159 Token = std::string(getTokenStr());
10161 if (getLoc() != Loc)
10166 if (!parseId(Suffix))
10170 StringRef DimId = Token;
10185 SMLoc S = getLoc();
10191 SMLoc Loc = getLoc();
10192 if (!parseDimId(Encoding))
10193 return Error(Loc,
"invalid dim value");
10196 AMDGPUOperand::CreateImm(
this, Encoding, S, AMDGPUOperand::ImmTyDim));
10205 SMLoc S = getLoc();
10214 if (!skipToken(
AsmToken::LBrac,
"expected an opening square bracket"))
10217 for (
size_t i = 0; i < 8; ++i) {
10221 SMLoc Loc = getLoc();
10222 if (getParser().parseAbsoluteExpression(Sels[i]))
10224 if (0 > Sels[i] || 7 < Sels[i])
10225 return Error(Loc,
"expected a 3-bit value");
10228 if (!skipToken(
AsmToken::RBrac,
"expected a closing square bracket"))
10232 for (
size_t i = 0; i < 8; ++i)
10233 DPP8 |= (Sels[i] << (i * 3));
10236 AMDGPUOperand::CreateImm(
this, DPP8, S, AMDGPUOperand::ImmTyDPP8));
10240bool AMDGPUAsmParser::isSupportedDPPCtrl(StringRef Ctrl,
10242 if (Ctrl ==
"row_newbcast")
10245 if (Ctrl ==
"row_share" || Ctrl ==
"row_xmask")
10248 if (Ctrl ==
"wave_shl" || Ctrl ==
"wave_shr" || Ctrl ==
"wave_rol" ||
10249 Ctrl ==
"wave_ror" || Ctrl ==
"row_bcast")
10252 return Ctrl ==
"row_mirror" ||
Ctrl ==
"row_half_mirror" ||
10253 Ctrl ==
"quad_perm" ||
Ctrl ==
"row_shl" ||
Ctrl ==
"row_shr" ||
10257int64_t AMDGPUAsmParser::parseDPPCtrlPerm() {
10260 if (!skipToken(
AsmToken::LBrac,
"expected an opening square bracket"))
10264 for (
int i = 0; i < 4; ++i) {
10269 SMLoc Loc = getLoc();
10270 if (getParser().parseAbsoluteExpression(Temp))
10272 if (Temp < 0 || Temp > 3) {
10273 Error(Loc,
"expected a 2-bit value");
10277 Val += (Temp << i * 2);
10280 if (!skipToken(
AsmToken::RBrac,
"expected a closing square bracket"))
10286int64_t AMDGPUAsmParser::parseDPPCtrlSel(StringRef Ctrl) {
10287 using namespace AMDGPU::DPP;
10292 SMLoc Loc = getLoc();
10294 if (getParser().parseAbsoluteExpression(Val))
10297 struct DppCtrlCheck {
10303 DppCtrlCheck
Check =
10304 StringSwitch<DppCtrlCheck>(Ctrl)
10305 .Case(
"wave_shl", {DppCtrl::WAVE_SHL1, 1, 1})
10306 .Case(
"wave_rol", {DppCtrl::WAVE_ROL1, 1, 1})
10307 .Case(
"wave_shr", {DppCtrl::WAVE_SHR1, 1, 1})
10308 .Case(
"wave_ror", {DppCtrl::WAVE_ROR1, 1, 1})
10309 .Case(
"row_shl", {DppCtrl::ROW_SHL0, 1, 15})
10310 .Case(
"row_shr", {DppCtrl::ROW_SHR0, 1, 15})
10311 .Case(
"row_ror", {DppCtrl::ROW_ROR0, 1, 15})
10312 .Case(
"row_share", {DppCtrl::ROW_SHARE_FIRST, 0, 15})
10313 .Case(
"row_xmask", {DppCtrl::ROW_XMASK_FIRST, 0, 15})
10314 .Case(
"row_newbcast", {DppCtrl::ROW_NEWBCAST_FIRST, 0, 15})
10318 if (
Check.Ctrl == -1) {
10319 Valid = (
Ctrl ==
"row_bcast" && (Val == 15 || Val == 31));
10327 Error(Loc, Twine(
"invalid ", Ctrl) + Twine(
" value"));
10335 using namespace AMDGPU::DPP;
10338 !isSupportedDPPCtrl(getTokenStr(),
Operands))
10341 SMLoc S = getLoc();
10347 if (Ctrl ==
"row_mirror") {
10348 Val = DppCtrl::ROW_MIRROR;
10349 }
else if (Ctrl ==
"row_half_mirror") {
10350 Val = DppCtrl::ROW_HALF_MIRROR;
10353 if (Ctrl ==
"quad_perm") {
10354 Val = parseDPPCtrlPerm();
10356 Val = parseDPPCtrlSel(Ctrl);
10365 AMDGPUOperand::CreateImm(
this, Val, S, AMDGPUOperand::ImmTyDppCtrl));
10371 OptionalImmIndexMap OptionalIdx;
10378 int OldIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::old);
10380 AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::src2_modifiers);
10381 bool IsMAC = OldIdx != -1 && Src2ModIdx != -1 &&
10385 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10386 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10390 int VdstInIdx = AMDGPU::getNamedOperandIdx(
Opc, AMDGPU::OpName::vdst_in);
10391 bool IsVOP3CvtSrDpp =
Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx12 ||
10392 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp8_gfx13 ||
10393 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx12 ||
10394 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp8_gfx13 ||
10395 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx12 ||
10396 Opc == AMDGPU::V_CVT_SR_BF8_F32_gfx12_e64_dpp_gfx13 ||
10397 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx12 ||
10398 Opc == AMDGPU::V_CVT_SR_FP8_F32_gfx12_e64_dpp_gfx13;
10404 if (OldIdx == NumOperands) {
10406 constexpr int DST_IDX = 0;
10408 }
else if (Src2ModIdx == NumOperands) {
10418 if (IsVOP3CvtSrDpp) {
10427 if (TiedTo != -1) {
10432 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10434 if (IsDPP8 &&
Op.isDppFI()) {
10437 Op.addRegOrImmWithFPInputModsOperands(Inst, 2);
10438 }
else if (
Op.isReg()) {
10439 Op.addRegOperands(Inst, 1);
10440 }
else if (
Op.isImm() &&
10442 Op.addImmOperands(Inst, 1);
10443 }
else if (
Op.isImm()) {
10444 OptionalIdx[
Op.getImmTy()] =
I;
10452 AMDGPUOperand::ImmTyClamp);
10458 AMDGPUOperand::ImmTyByteSel);
10463 AMDGPUOperand::ImmTyOModSI);
10466 cvtVOP3P(Inst,
Operands, OptionalIdx);
10468 cvtVOP3OpSel(Inst,
Operands, OptionalIdx);
10471 AMDGPUOperand::ImmTyOpSel);
10476 AMDGPUOperand::ImmTyDPP8);
10477 using namespace llvm::AMDGPU::DPP;
10481 AMDGPUOperand::ImmTyDppCtrl, 0xe4);
10483 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10485 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10487 AMDGPUOperand::ImmTyDppBoundCtrl);
10491 AMDGPUOperand::ImmTyDppFI);
10497 OptionalImmIndexMap OptionalIdx;
10501 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10502 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10509 if (TiedTo != -1) {
10514 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10516 if (
Op.isReg() && validateVccOperand(
Op.getReg())) {
10524 Op.addImmOperands(Inst, 1);
10526 Op.addRegWithFPInputModsOperands(Inst, 2);
10527 }
else if (
Op.isDppFI()) {
10529 }
else if (
Op.isReg()) {
10530 Op.addRegOperands(Inst, 1);
10536 Op.addRegWithFPInputModsOperands(Inst, 2);
10537 }
else if (
Op.isReg()) {
10538 Op.addRegOperands(Inst, 1);
10539 }
else if (
Op.isDPPCtrl()) {
10540 Op.addImmOperands(Inst, 1);
10541 }
else if (
Op.isImm()) {
10543 OptionalIdx[
Op.getImmTy()] =
I;
10551 using namespace llvm::AMDGPU::DPP;
10555 AMDGPUOperand::ImmTyDppRowMask, 0xf);
10557 AMDGPUOperand::ImmTyDppBankMask, 0xf);
10559 AMDGPUOperand::ImmTyDppBoundCtrl);
10562 AMDGPUOperand::ImmTyDppFI);
10573 AMDGPUOperand::ImmTy
Type) {
10574 return parseStringOrIntWithPrefix(
10576 {
"BYTE_0",
"BYTE_1",
"BYTE_2",
"BYTE_3",
"WORD_0",
"WORD_1",
"DWORD"},
10581 return parseStringOrIntWithPrefix(
10582 Operands,
"dst_unused", {
"UNUSED_PAD",
"UNUSED_SEXT",
"UNUSED_PRESERVE"},
10583 AMDGPUOperand::ImmTySDWADstUnused);
10587 cvtSDWA(Inst,
Operands, SDWAInstType::VOP1);
10591 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2);
10594void AMDGPUAsmParser::cvtSdwaVOP2b(MCInst &Inst,
10596 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2,
true,
true);
10599void AMDGPUAsmParser::cvtSdwaVOP2e(MCInst &Inst,
10601 cvtSDWA(Inst,
Operands, SDWAInstType::VOP2,
false,
true);
10609 SDWAInstType BasicInstType,
bool SkipDstVcc,
10611 using namespace llvm::AMDGPU::SDWA;
10613 OptionalImmIndexMap OptionalIdx;
10614 bool SkipVcc = SkipDstVcc || SkipSrcVcc;
10615 bool SkippedVcc =
false;
10619 for (
unsigned J = 0; J <
Desc.getNumDefs(); ++J) {
10620 ((AMDGPUOperand &)*
Operands[
I++]).addRegOperands(Inst, 1);
10624 AMDGPUOperand &
Op = ((AMDGPUOperand &)*
Operands[
I]);
10625 if (SkipVcc && !SkippedVcc &&
Op.isReg() &&
10626 (
Op.getReg() == AMDGPU::VCC ||
Op.getReg() == AMDGPU::VCC_LO)) {
10632 if (BasicInstType == SDWAInstType::VOP2 &&
10638 if (BasicInstType == SDWAInstType::VOPC && Inst.
getNumOperands() == 0) {
10644 Op.addRegOrImmWithInputModsOperands(Inst, 2);
10645 }
else if (
Op.isImm()) {
10647 OptionalIdx[
Op.getImmTy()] =
I;
10651 SkippedVcc =
false;
10655 if (
Opc != AMDGPU::V_NOP_sdwa_gfx10 &&
Opc != AMDGPU::V_NOP_sdwa_gfx9 &&
10656 Opc != AMDGPU::V_NOP_sdwa_vi) {
10658 switch (BasicInstType) {
10659 case SDWAInstType::VOP1:
10662 AMDGPUOperand::ImmTyClamp, 0);
10666 AMDGPUOperand::ImmTyOModSI, 0);
10670 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10674 AMDGPUOperand::ImmTySDWADstUnused,
10675 DstUnused::UNUSED_PRESERVE);
10678 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10681 case SDWAInstType::VOP2:
10683 AMDGPUOperand::ImmTyClamp, 0);
10687 AMDGPUOperand::ImmTyOModSI, 0);
10690 AMDGPUOperand::ImmTySDWADstSel, SdwaSel::DWORD);
10692 AMDGPUOperand::ImmTySDWADstUnused,
10693 DstUnused::UNUSED_PRESERVE);
10695 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10697 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10700 case SDWAInstType::VOPC:
10703 AMDGPUOperand::ImmTyClamp, 0);
10705 AMDGPUOperand::ImmTySDWASrc0Sel, SdwaSel::DWORD);
10707 AMDGPUOperand::ImmTySDWASrc1Sel, SdwaSel::DWORD);
10714 if (Inst.
getOpcode() == AMDGPU::V_MAC_F32_sdwa_vi ||
10715 Inst.
getOpcode() == AMDGPU::V_MAC_F16_sdwa_vi) {
10716 auto *it = Inst.
begin();
10718 it, AMDGPU::getNamedOperandIdx(Inst.
getOpcode(), AMDGPU::OpName::src2));
10731#define GET_MATCHER_IMPLEMENTATION
10732#define GET_MNEMONIC_SPELL_CHECKER
10733#define GET_MNEMONIC_CHECKER
10734#include "AMDGPUGenAsmMatcher.inc"
10740 return parseTokenOp(
"addr64",
Operands);
10742 return parseNamedBit(
"done",
Operands, AMDGPUOperand::ImmTyDone,
true);
10744 return parseTokenOp(
"idxen",
Operands);
10746 return parseNamedBit(
"lds",
Operands, AMDGPUOperand::ImmTyLDS,
10749 return parseTokenOp(
"offen",
Operands);
10751 return parseTokenOp(
"off",
Operands);
10752 case MCK_row_95_en:
10753 return parseNamedBit(
"row_en",
Operands, AMDGPUOperand::ImmTyRowEn,
true);
10755 return parseNamedBit(
"gds",
Operands, AMDGPUOperand::ImmTyGDS);
10757 return parseNamedBit(
"tfe",
Operands, AMDGPUOperand::ImmTyTFE);
10759 return tryCustomParseOperand(
Operands, MCK);
10764unsigned AMDGPUAsmParser::validateTargetOperandClass(MCParsedAsmOperand &
Op,
10770 AMDGPUOperand &Operand = (AMDGPUOperand &)
Op;
10773 return Operand.isAddr64() ? Match_Success : Match_InvalidOperand;
10775 return Operand.isGDS() ? Match_Success : Match_InvalidOperand;
10777 return Operand.isLDS() ? Match_Success : Match_InvalidOperand;
10779 return Operand.isIdxen() ? Match_Success : Match_InvalidOperand;
10781 return Operand.isOffen() ? Match_Success : Match_InvalidOperand;
10783 return Operand.isTFE() ? Match_Success : Match_InvalidOperand;
10785 return Operand.isDone() ? Match_Success : Match_InvalidOperand;
10786 case MCK_row_95_en:
10787 return Operand.isRowEn() ? Match_Success : Match_InvalidOperand;
10795 return Operand.isSSrc_b32() ? Match_Success : Match_InvalidOperand;
10797 return Operand.isSSrc_f32() ? Match_Success : Match_InvalidOperand;
10798 case MCK_SOPPBrTarget:
10799 return Operand.isSOPPBrTarget() ? Match_Success : Match_InvalidOperand;
10800 case MCK_VReg32OrOff:
10801 return Operand.isVReg32OrOff() ? Match_Success : Match_InvalidOperand;
10802 case MCK_InterpSlot:
10803 return Operand.isInterpSlot() ? Match_Success : Match_InvalidOperand;
10804 case MCK_InterpAttr:
10805 return Operand.isInterpAttr() ? Match_Success : Match_InvalidOperand;
10806 case MCK_InterpAttrChan:
10807 return Operand.isInterpAttrChan() ? Match_Success : Match_InvalidOperand;
10809 case MCK_SReg_64_XEXEC:
10819 return Operand.isNull() ? Match_Success : Match_InvalidOperand;
10821 return Match_InvalidOperand;
10830 SMLoc S = getLoc();
10839 return Error(S,
"expected a 16-bit value");
10842 AMDGPUOperand::CreateImm(
this,
Imm, S, AMDGPUOperand::ImmTyEndpgm));
10846bool AMDGPUOperand::isEndpgm()
const {
return isImmTy(ImmTyEndpgm); }
10852bool AMDGPUOperand::isSplitBarrier()
const {
static const TargetRegisterClass * getRegClass(const MachineInstr &MI, Register Reg)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
SmallVector< int16_t, MAX_SRC_OPERANDS_NUM > OperandIndices
static bool checkWriteLane(const MCInst &Inst)
static bool getRegNum(StringRef Str, unsigned &Num)
static void addSrcModifiersAndSrc(MCInst &Inst, const OperandVector &Operands, unsigned i, unsigned Opc, AMDGPU::OpName OpName)
static constexpr RegInfo RegularRegisters[]
static const RegInfo * getRegularRegInfo(StringRef Str)
static ArrayRef< unsigned > getAllVariants()
static OperandIndices getSrcOperandIndices(unsigned Opcode, bool AddMandatoryLiterals=false)
static int IsAGPROperand(const MCInst &Inst, AMDGPU::OpName Name, const MCRegisterInfo *MRI)
static bool IsMovrelsSDWAOpcode(const unsigned Opcode)
static const fltSemantics * getFltSemantics(unsigned Size)
static bool isRegularReg(RegisterKind Kind)
LLVM_ABI LLVM_EXTERNAL_VISIBILITY void LLVMInitializeAMDGPUAsmParser()
Force static initialization.
static bool ConvertOmodMul(int64_t &Mul)
#define PARSE_BITS_ENTRY(FIELD, ENTRY, VALUE, RANGE)
static bool isInlineableLiteralOp16(int64_t Val, MVT VT, bool HasInv2Pi)
static bool canLosslesslyConvertToFPType(APFloat &FPLiteral, MVT VT)
static bool AMDGPUCheckMnemonic(StringRef Mnemonic, const FeatureBitset &AvailableFeatures, unsigned VariantID)
static void applyMnemonicAliases(StringRef &Mnemonic, const FeatureBitset &Features, unsigned VariantID)
constexpr unsigned MAX_SRC_OPERANDS_NUM
#define EXPR_RESOLVE_OR_ERROR(RESOLVED)
static bool ConvertOmodDiv(int64_t &Div)
static bool IsRevOpcode(const unsigned Opcode)
static bool encodeCnt(const AMDGPU::IsaVersion ISA, int64_t &IntVal, int64_t CntVal, bool Saturate, unsigned(*encode)(const IsaVersion &Version, unsigned, unsigned), unsigned(*decode)(const IsaVersion &Version, unsigned))
static MCRegister getSpecialRegForName(StringRef RegName)
static void addOptionalImmOperand(MCInst &Inst, const OperandVector &Operands, AMDGPUAsmParser::OptionalImmIndexMap &OptionalIdx, AMDGPUOperand::ImmTy ImmT, int64_t Default=0, std::optional< unsigned > InsertAt=std::nullopt)
static void cvtVOP3DstOpSelOnly(MCInst &Inst, const MCRegisterInfo &MRI)
static bool isRegOrImmWithInputMods(const MCInstrDesc &Desc, unsigned OpNum)
static const fltSemantics * getOpFltSemantics(uint8_t OperandType)
static bool isInvalidVOPDY(const OperandVector &Operands, uint64_t InvalidOprIdx)
static std::string AMDGPUMnemonicSpellCheck(StringRef S, const FeatureBitset &FBS, unsigned VariantID=0)
static LLVM_READNONE unsigned encodeBitmaskPerm(const unsigned AndMask, const unsigned OrMask, const unsigned XorMask)
static bool isSafeTruncation(int64_t Val, unsigned Size)
AMDHSA kernel descriptor MCExpr struct for use in MC layer.
Provides AMDGPU specific target descriptions.
Enums shared between the AMDGPU backend (LLVM) and the ELF linker (LLD) for the .amdgpu....
AMDHSA kernel descriptor definitions.
static bool parseExpr(MCAsmParser &MCParser, const MCExpr *&Value, raw_ostream &Err)
MC layer struct for AMDGPUMCKernelCodeT, provides MCExpr functionality where required.
@ AMD_CODE_PROPERTY_ENABLE_WAVEFRONT_SIZE32
This file declares a class to represent arbitrary precision floating point values and provide a varie...
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_EXTERNAL_VISIBILITY
static llvm::Expected< InlineInfo > decode(GsymDataExtractor &Data, uint64_t &Offset, uint64_t BaseAddr)
Decode an InlineInfo in Data at the specified offset.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
Loop::LoopBounds::Direction Direction
static bool hasFeature(StringRef Feature, const FeatureBitset &FeatureBits, ArrayRef< SubtargetFeatureKV > ProcFeatures)
Register const TargetRegisterInfo * TRI
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
static bool isReg(const MCInst &MI, unsigned OpNo)
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t IntrinsicInst * II
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
Interface definition for SIInstrInfo.
Func getContext().diagnose(DiagnosticInfoUnsupported(Func
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
This file implements the SmallBitVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static void initialize(TargetLibraryInfoImpl &TLI, const Triple &T, const llvm::StringTable &StandardNames, VectorLibrary VecLib)
Initialize the set of available library functions based on the specified target triple.
static const char * getRegisterName(MCRegister Reg)
static const AMDGPUMCExpr * createMax(ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static unsigned getNumExpectedArgs(VariantKind Kind)
static const AMDGPUMCExpr * createLit(LitModifier Lit, int64_t Value, MCContext &Ctx)
static const AMDGPUMCExpr * create(VariantKind Kind, ArrayRef< const MCExpr * > Args, MCContext &Ctx)
static const AMDGPUMCExpr * createExtraSGPRs(const MCExpr *VCCUsed, const MCExpr *FlatScrUsed, bool XNACKUsed, MCContext &Ctx)
Allow delayed MCExpr resolve of ExtraSGPRs (in case VCCUsed or FlatScrUsed are unresolvable but neede...
static const AMDGPUMCExpr * createAlignTo(const MCExpr *Value, const MCExpr *Align, MCContext &Ctx)
static std::optional< TargetID > parseTargetIDString(StringRef TargetIDDirective)
Parse and validate a TargetID from a full "<triple>-<processor>:<features>" directive string.
TargetIDSetting getXnackSetting() const
GPUKind getGPUKind() const
StringRef getTargetTripleString() const
std::string toString() const
TargetIDSetting getSramEccSetting() const
static const fltSemantics & IEEEsingle()
static const fltSemantics & BFloat()
static const fltSemantics & IEEEdouble()
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
opStatus
IEEE-754R 7: Default exception handling.
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
ArrayRef< T > drop_front(size_t N=1) const
Drop the first N elements of the array.
const T & front() const
Get the first element.
size_t size() const
Get the array size.
bool empty() const
Check if the array is empty.
StringRef getString() const
Get the string for the current token, this includes all characters (for example, the quotes on string...
bool is(TokenKind K) const
Container class for subtarget features.
constexpr bool test(unsigned I) const
constexpr FeatureBitset & flip(unsigned I)
void printExpr(raw_ostream &, const MCExpr &) const
virtual void Initialize(MCAsmParser &Parser)
Initialize the extension for parsing using the given Parser.
static const MCBinaryExpr * createAdd(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx, SMLoc Loc=SMLoc())
static const MCBinaryExpr * createDiv(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static const MCBinaryExpr * createSub(const MCExpr *LHS, const MCExpr *RHS, MCContext &Ctx)
static LLVM_ABI const MCConstantExpr * create(int64_t Value, MCContext &Ctx, bool PrintInHex=false, unsigned SizeInBytes=0)
Context object for machine code objects.
LLVM_ABI MCSymbol * getOrCreateSymbol(const Twine &Name)
Lookup the symbol inside with the specified Name.
Instances of this class represent a single low-level machine instruction.
unsigned getNumOperands() const
unsigned getOpcode() const
iterator insert(iterator I, const MCOperand &Op)
void addOperand(const MCOperand Op)
const MCOperand & getOperand(unsigned i) const
Describe properties that are true of each instruction in the target description file.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
const int16_t * getRegClassByHwModeTable(unsigned ModeId) const
int16_t getOpRegClassID(const MCOperandInfo &OpInfo, unsigned HwModeId) const
Return the ID of the register class to use for OpInfo, for the active HwMode HwModeId.
Instances of this class represent operands of the MCInst class.
static MCOperand createExpr(const MCExpr *Val)
static MCOperand createReg(MCRegister Reg)
static MCOperand createImm(int64_t Val)
void setReg(MCRegister Reg)
Set the register number.
MCRegister getReg() const
Returns the register number.
const MCExpr * getExpr() const
MCParsedAsmOperand - This abstract class represents a source-level assembly instruction operand.
MCRegisterClass - Base class of TargetRegisterClass.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
bool contains(MCRegister Reg) const
contains - Return true if the specified register is included in this register class.
MCRegisterInfo base class - We assume that the target defines a static array of MCRegisterDesc object...
bool regsOverlap(MCRegister RegA, MCRegister RegB) const
Returns true if the two registers are equal or alias each other.
const MCRegisterClass & getRegClass(unsigned i) const
Returns the register class associated with the enumeration value.
MCRegister getSubReg(MCRegister Reg, unsigned Idx) const
Returns the physical register number of sub-register "Index" for physical register RegNo.
Wrapper class representing physical registers. Should be passed by value.
constexpr bool isValid() const
Generic base class for all target subtargets.
MCSymbol - Instances of this class represent a symbol name in the MC file, and MCSymbols are created ...
StringRef getName() const
getName - Get the symbol name.
bool isVariable() const
isVariable - Check if this is a variable symbol.
LLVM_ABI void setVariableValue(const MCExpr *Value)
void setRedefinable(bool Value)
Mark this symbol as redefinable.
const MCExpr * getVariableValue() const
Get the expression of the variable symbol.
MCTargetAsmParser - Generic interface to target specific assembly parsers.
uint64_t getScalarSizeInBits() const
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
Ternary parse status returned by various parse* methods.
constexpr bool isFailure() const
static constexpr StatusTy Failure
constexpr bool isSuccess() const
static constexpr StatusTy Success
static constexpr StatusTy NoMatch
constexpr bool isNoMatch() const
constexpr unsigned id() const
Represents a location in source code.
static SMLoc getFromPointer(const char *Ptr)
constexpr const char * getPointer() const
constexpr bool isValid() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
bool contains(ConstPtrType Ptr) const
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
A wrapper around a string literal that serves as a proxy for constructing global tables of StringRefs...
Represent a constant reference to a string, i.e.
bool consume_back(StringRef Suffix)
Returns true if this StringRef has the given suffix and removes that suffix.
constexpr StringRef substr(size_t Start, size_t N=npos) const
Return a reference to the substring from [Start, Start + N).
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
constexpr bool empty() const
Check if the string is empty.
StringRef drop_front(size_t N=1) const
Return a StringRef equal to 'this' but with the first N elements dropped.
constexpr size_t size() const
Get the string size.
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
bool ends_with(StringRef Suffix) const
Check if this string ends with the given Suffix.
bool consume_front(char Prefix)
Returns true if this StringRef has the given prefix and removes that prefix.
bool contains(StringRef key) const
Check if the set contains the given key.
std::pair< typename Base::iterator, bool > insert(StringRef key)
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
LLVM_ABI std::string str() const
Return the twine contents as a std::string.
std::pair< iterator, bool > insert(const ValueT &V)
This class implements an extremely fast bulk output stream that can only output to a stream.
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
int encodeDepCtr(const StringRef Name, int64_t Val, unsigned &UsedOprMask, const MCSubtargetInfo &STI)
int getDefaultDepCtrEncoding(const MCSubtargetInfo &STI)
bool isSupportedTgtId(unsigned Id, const MCSubtargetInfo &STI)
unsigned getTgtId(const StringRef Name)
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumSGPRs[]
Key for Kernel::CodeProps::Metadata::mNumSGPRs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
constexpr char AssemblerDirectiveBegin[]
HSA metadata beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
HSA metadata ending assembler directive.
constexpr char AssemblerDirectiveBegin[]
Old HSA metadata beginning assembler directive for V2.
int64_t getHwregId(StringRef Name, const MCSubtargetInfo &STI)
@ FIXED_NUM_SGPRS_FOR_INIT_BUG
unsigned getVGPREncodingGranule(const MCSubtargetInfo &STI, std::optional< bool > EnableWavefrontSize32)
unsigned getSGPREncodingGranule(const MCSubtargetInfo &STI)
bool targetIDSettingsConflict(TargetIDSetting Lhs, TargetIDSetting Rhs)
Returns true if Lhs and Rhs are incompatible (both specific but different).
unsigned getLocalMemorySize(const MCSubtargetInfo &STI)
constexpr char AssemblerDirective[]
PAL metadata (old linear format) assembler directive.
constexpr char AssemblerDirectiveBegin[]
PAL metadata (new MsgPack format) beginning assembler directive.
constexpr char AssemblerDirectiveEnd[]
PAL metadata (new MsgPack format) ending assembler directive.
int64_t getMsgOpId(int64_t MsgId, StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a sendmsg operation to the operation portion of the immediate encoding.
int64_t getMsgId(StringRef Name, const MCSubtargetInfo &STI)
Map from a symbolic name for a msg_id to the message portion of the immediate encoding.
uint64_t encodeMsg(uint64_t MsgId, uint64_t OpId, uint64_t StreamId)
bool msgSupportsStream(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI)
bool isValidMsgId(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgStream(int64_t MsgId, int64_t OpId, int64_t StreamId, const MCSubtargetInfo &STI, bool Strict)
bool msgRequiresOp(int64_t MsgId, const MCSubtargetInfo &STI)
bool isValidMsgOp(int64_t MsgId, int64_t OpId, const MCSubtargetInfo &STI, bool Strict)
ArrayRef< GFXVersion > getGFXVersions()
constexpr unsigned COMPONENTS[]
constexpr const char *const ModMatrixFmt[]
constexpr const char *const ModMatrixScaleFmt[]
constexpr const char *const ModMatrixScale[]
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
bool isInlineValue(MCRegister Reg)
bool isPKFMACF16InlineConstant(uint32_t Literal, bool IsGFX11Plus)
LLVM_READONLY const MIMGInfo * getMIMGInfo(unsigned Opc)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
bool isSGPR(MCRegister Reg, const MCRegisterInfo *TRI)
Is Reg - scalar register.
MCRegister getMCReg(MCRegister Reg, const MCSubtargetInfo &STI)
If Reg is a pseudo reg, return the correct hardware register given STI otherwise return Reg.
FuncInfoFlags
Per-function flags packed into INFO_FLAGS entries.
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const int OPR_ID_UNSUPPORTED
bool isInlinableLiteralV2I16(uint32_t Literal)
bool isHi16Reg(MCRegister Reg, const MCRegisterInfo &MRI)
unsigned getTemporalHintType(const MCInstrDesc TID)
LLVM_READONLY bool isLitExpr(const MCExpr *Expr)
bool isInlinableLiteralV2BF16(uint32_t Literal)
LLVM_ABI bool isCPUValidForSubArch(Triple::SubArchType SubArch, GPUKind AK)
Return true if the GPU AK is usable with the triple subarch SubArch.
unsigned getMaxNumUserSGPRs(const MCSubtargetInfo &STI)
unsigned getNumFlatOffsetBits(const MCSubtargetInfo &ST)
For pre-GFX12 FLAT instructions the offset must be positive; MSB is ignored and forced to zero.
bool hasA16(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedSignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset, bool IsBuffer)
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
bool hasPackedD16(const MCSubtargetInfo &STI)
bool isGFX940(const MCSubtargetInfo &STI)
bool isInlinableLiteralV2F16(uint32_t Literal)
bool isHsaAbi(const MCSubtargetInfo &STI)
bool isGFX11(const MCSubtargetInfo &STI)
const int OPR_VAL_INVALID
bool getSMEMIsBuffer(unsigned Opc)
bool isPackedSingleSGPRFP32Inst(unsigned Opc)
The opcode is a packed fp32 instruction which only reads low 32 bits of a scalar operand and propagat...
LLVM_ABI unsigned getAddressableNumSGPRs(GPUKind AK)
uint8_t mfmaScaleF8F6F4FormatToNumRegs(unsigned EncodingVal)
LLVM_ABI IsaVersion getIsaVersion(StringRef GPU)
unsigned getRegBitWidth(unsigned RCID)
Get the size in bits of a register from the register class RC.
bool isValid32BitLiteral(uint64_t Val, bool IsFP64)
LLVM_ABI unsigned getTotalNumVGPRs(GPUKind AK, bool IsWave32)
CanBeVOPD getCanBeVOPD(unsigned Opc, unsigned EncodingFamily, bool VOPD3)
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
bool isSI(const MCSubtargetInfo &STI)
bool hasPrivateApertureRegs(const MCSubtargetInfo &STI)
unsigned decodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt)
unsigned getWaitcntBitMask(const IsaVersion &Version)
bool isGFX9(const MCSubtargetInfo &STI)
unsigned getVOPDEncodingFamily(const MCSubtargetInfo &ST)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
bool isKImmOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this a KImm operand?
GPUKind
GPU kinds supported by the AMDGPU target.
bool isGFX90A(const MCSubtargetInfo &STI)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByEncoding(uint8_t DimEnc)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
bool isGFX12(const MCSubtargetInfo &STI)
unsigned encodeExpcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Expcnt)
bool hasMAIInsts(const MCSubtargetInfo &STI)
constexpr bool isSISrcOperand(const MCOperandInfo &OpInfo)
Is this an AMDGPU specific source operand?
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfoByAsmSuffix(StringRef AsmSuffix)
bool hasMIMG_R128(const MCSubtargetInfo &STI)
LLVM_ABI GPUKind parseArchAMDGCN(StringRef CPU)
bool hasG16(const MCSubtargetInfo &STI)
unsigned getAddrSizeMIMGOp(const MIMGBaseOpcodeInfo *BaseOpcode, const MIMGDimInfo *Dim, bool IsA16, bool IsG16Supported)
bool hasArchitectedFlatScratch(const MCSubtargetInfo &STI)
LLVM_READONLY int64_t getLitValue(const MCExpr *Expr)
bool isGFX11Plus(const MCSubtargetInfo &STI)
bool isSISrcFPOperand(const MCInstrDesc &Desc, unsigned OpNo)
Is this floating-point operand?
bool isGFX10Plus(const MCSubtargetInfo &STI)
AMDGPU::TargetID TargetID
int64_t encode32BitLiteral(int64_t Imm, OperandType Type, bool IsLit)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
@ OPERAND_KIMM32
Operand with 32-bit immediate that uses the constant bus.
@ OPERAND_REG_INLINE_C_FP64
@ OPERAND_REG_IMM_NOINLINE_FP16
@ OPERAND_REG_INLINE_C_BF16
@ OPERAND_REG_INLINE_C_V2BF16
@ OPERAND_REG_IMM_V2INT64
@ OPERAND_REG_IMM_V2INT16
@ OPERAND_REG_IMM_INT32
Operands with register, 32-bit, or 64-bit immediate.
@ OPERAND_REG_IMM_V2FP16_SPLAT
@ OPERAND_REG_INLINE_C_INT64
@ OPERAND_REG_INLINE_C_INT16
Operands with register or inline constant.
@ OPERAND_REG_IMM_NOINLINE_V2FP16
@ OPERAND_REG_INLINE_C_V2FP16
@ OPERAND_REG_INLINE_AC_INT32
Operands with an AccVGPR register or inline constant.
@ OPERAND_REG_INLINE_AC_FP32
@ OPERAND_REG_IMM_V2INT32
@ OPERAND_REG_INLINE_C_FP32
@ OPERAND_REG_INLINE_C_INT32
@ OPERAND_REG_INLINE_C_V2INT16
@ OPERAND_REG_INLINE_AC_FP64
@ OPERAND_REG_INLINE_C_FP16
@ OPERAND_INLINE_SPLIT_BARRIER_INT32
constexpr bool isBF16SrcOperand(const MCOperandInfo &OpInfo)
Is this a scalar (i.e. not packed) bf16 source operand?
bool isDPALU_DPP(const MCInstrDesc &OpDesc, const MCInstrInfo &MII, const MCSubtargetInfo &ST)
LLVM_ABI StringRef getArchNameAMDGCN(GPUKind AK)
bool hasGDS(const MCSubtargetInfo &STI)
bool isLegalSMRDEncodedUnsignedOffset(const MCSubtargetInfo &ST, int64_t EncodedOffset)
bool isGFX9Plus(const MCSubtargetInfo &STI)
bool hasDPPSrc1SGPR(const MCSubtargetInfo &STI)
const int OPR_ID_DUPLICATE
bool isVOPD(unsigned Opc)
VOPD::InstInfo getVOPDInstInfo(const MCInstrDesc &OpX, const MCInstrDesc &OpY)
unsigned encodeVmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Vmcnt)
unsigned decodeExpcnt(const IsaVersion &Version, unsigned Waitcnt)
const MIMGBaseOpcodeInfo * getMIMGBaseOpcode(unsigned Opc)
bool isVI(const MCSubtargetInfo &STI)
bool supportsScaleOffset(const MCInstrInfo &MII, unsigned Opcode)
MCRegister mc2PseudoReg(MCRegister Reg)
Convert hardware register Reg to a pseudo register.
unsigned hasKernargPreload(const MCSubtargetInfo &STI)
bool supportsWGP(const MCSubtargetInfo &STI)
LLVM_READNONE unsigned getOperandSize(const MCOperandInfo &OpInfo)
bool isCI(const MCSubtargetInfo &STI)
unsigned encodeLgkmcnt(const IsaVersion &Version, unsigned Waitcnt, unsigned Lgkmcnt)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
bool isGFX1250Plus(const MCSubtargetInfo &STI)
bool hasPopsExitingWaveID(const MCSubtargetInfo &STI)
unsigned decodeVmcnt(const IsaVersion &Version, unsigned Waitcnt)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
bool hasVOPD(const MCSubtargetInfo &STI)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
bool isPermlane16(unsigned Opc)
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ UNDEF
UNDEF - An undefined node.
Predicate getPredicate(unsigned Condition, unsigned Hint)
Return predicate consisting of specified condition and hint bits.
void validate(const Triple &TT, const FeatureBitset &FeatureBits)
constexpr bool isAtomicRet(const T &...O)
constexpr bool isVOPC(const T &...O)
constexpr bool isVOP3(const T &...O)
constexpr bool isVOP1(const T &...O)
constexpr bool usesTENSOR_CNT(const T &...O)
constexpr bool isMAI(const T &...O)
constexpr bool isVOP2(const T &...O)
constexpr bool isSWMMAC(const T &...O)
constexpr bool isSOP2(const T &...O)
constexpr bool isFLAT(const T &...O)
constexpr bool isVOP3P(const T &...O)
constexpr bool isBuffer(const T &...O)
constexpr bool hasIntClamp(const T &...O)
constexpr bool isAtomicNoRet(const T &...O)
constexpr bool isSMRD(const T &...O)
constexpr bool isVOP3Like(const T &...O)
constexpr bool isMIMG(const T &...O)
constexpr bool isVMEM(const T &...O)
constexpr bool isImage(const T &...O)
constexpr bool isWMMA(const T &...O)
constexpr bool isVOPD3(const T &...O)
constexpr bool isGWS(const T &...O)
constexpr bool isMUBUF(const T &...O)
constexpr bool isSDWA(const T &...O)
constexpr bool isSOPC(const T &...O)
constexpr bool isDOT(const T &...O)
constexpr bool isVSAMPLE(const T &...O)
constexpr bool isDS(const T &...O)
constexpr bool isAtomic(const T &...O)
constexpr bool isGather4(const T &...O)
constexpr bool isPacked(const T &...O)
constexpr bool isDPP(const T &...O)
constexpr bool isSegmentSpecificFLAT(const T &...O)
@ Valid
The data is already valid.
Scope
Defines the scope in which this symbol should be visible: Default – Visible in the public interface o...
EnumSet< Modifier > Modifiers
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
bool errorToBool(Error Err)
Helper for converting an Error to a bool.
StringMapEntry< Value * > ValueName
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Printable print(const GCNRegPressure &RP, const GCNSubtarget *ST=nullptr, unsigned DynamicVGPRBlockSize=0)
unsigned encode(MaybeAlign A)
Returns a representation of the alignment that encodes undefined as 0.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
static bool isMem(const MachineInstr &MI, unsigned Op)
LLVM_ABI std::pair< StringRef, StringRef > getToken(StringRef Source, StringRef Delimiters=" \t\n\v\f\r")
getToken - This function extracts one token from source, ignoring any leading characters that appear ...
static StringRef getCPU(StringRef CPU)
Processes a CPU name.
testing::Matcher< const detail::ErrorHolder & > Failed()
LLVM_ABI void PrintError(const Twine &Msg)
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Target & getTheR600Target()
The target for R600 GPUs.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
SmallVectorImpl< std::unique_ptr< MCParsedAsmOperand > > OperandVector
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
MutableArrayRef(T &OneElt) -> MutableArrayRef< T >
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Target & getTheGCNTarget()
The target for GCN GPUs.
@ Sub
Subtraction of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
DWARFExpression::Operation Op
raw_ostream & operator<<(raw_ostream &OS, const APFixedPoint &FX)
unsigned M0(unsigned Val)
ArrayRef(const T &OneElt) -> ArrayRef< T >
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Target & getTheGCNLegacyTarget()
The target for GCN GPUs, registered under the legacy "amdgcn" architecture name for use with -march.
@ Enabled
Convert any .debug_str_offsets tables to DWARF64 if needed.
void initDefault(const MCSubtargetInfo &STI, MCContext &Ctx, bool InitMCExpr=true)
void validate(const MCSubtargetInfo *STI, MCContext &Ctx)
uint32_t PrivateSegmentSize
SmallVector< std::pair< MCSymbol *, std::string >, 4 > IndirectCalls
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 8 > Calls
SmallVector< FuncInfo, 8 > Funcs
SmallVector< std::pair< MCSymbol *, std::string >, 4 > TypeIds
SmallVector< std::pair< MCSymbol *, MCSymbol * >, 4 > Uses
Instruction set architecture version.
const MCExpr * compute_pgm_rsrc2
const MCExpr * kernarg_size
const MCExpr * kernarg_preload
const MCExpr * compute_pgm_rsrc3
const MCExpr * private_segment_fixed_size
const MCExpr * compute_pgm_rsrc1
static void bits_set(const MCExpr *&Dst, const MCExpr *Value, uint32_t Shift, uint32_t Mask, MCContext &Ctx)
const MCExpr * group_segment_fixed_size
static MCKernelDescriptor getDefaultAmdhsaKernelDescriptor(const MCSubtargetInfo *STI, MCContext &Ctx)
const MCExpr * kernel_code_properties
RegisterMCAsmParser - Helper template for registering a target specific assembly parser,...
uint32_t group_segment_fixed_size
uint32_t private_segment_fixed_size