28#include "llvm/IR/IntrinsicsAMDGPU.h"
36#define DEBUG_TYPE "AMDGPUtti"
40struct AMDGPUImageDMaskIntrinsic {
44#define GET_AMDGPUImageDMaskIntrinsicTable_IMPL
45#include "AMDGPUGenSearchableTables.inc"
56 "nans handled separately");
74 bool AllowI16SExt =
false) {
75 Type *VTy = V.getType();
84 APFloat FloatValue(ConstFloat->getValueAPF());
85 bool LosesInfo =
true;
94 APInt IntValue(ConstInt->getValue());
103 Value *CastCandidate;
110 if (!IsExt && !IsFloat && AllowI16SExt)
123 Type *VTy = V.getType();
132 return Builder.CreateExtractElement(VecCast->
getOperand(0), Idx);
156 Func(Args, OverloadTys);
172 bool RemoveOldIntr = &OldIntr != &InstToReplace;
181static std::optional<Instruction *>
189 if (
const auto *LZMappingInfo =
191 if (
auto *ConstantLod =
193 if (ConstantLod->isZero() || ConstantLod->isNegative()) {
198 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
199 Args.erase(Args.begin() + ImageDimIntr->LodIndex);
206 if (
const auto *MIPMappingInfo =
208 if (
auto *ConstantMip =
210 if (ConstantMip->isZero()) {
215 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
216 Args.erase(Args.begin() + ImageDimIntr->MipIndex);
223 if (
const auto *BiasMappingInfo =
225 if (
auto *ConstantBias =
227 if (ConstantBias->isZero()) {
232 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
233 Args.erase(Args.begin() + ImageDimIntr->BiasIndex);
234 ArgTys.erase(ArgTys.begin() + ImageDimIntr->BiasTyArg);
241 if (
const auto *OffsetMappingInfo =
243 if (
auto *ConstantOffset =
245 if (ConstantOffset->isZero()) {
248 OffsetMappingInfo->NoOffset, ImageDimIntr->
Dim);
250 II,
II, NewImageDimIntr->
Intr, IC, [&](
auto &Args,
auto &ArgTys) {
251 Args.erase(Args.begin() + ImageDimIntr->OffsetIndex);
267 (DimInfo->
MSAA ? 1 : 0);
269 if (ConstantSlice && ConstantSlice->isZero()) {
274 [&](
auto &Args,
auto &ArgTys) {
275 Args.erase(Args.begin() + SliceIndex);
282 if (ST->hasD16Images()) {
288 if (
II.hasOneUse()) {
291 if (
User->getOpcode() == Instruction::FPTrunc &&
295 [&](
auto &Args,
auto &ArgTys) {
298 ArgTys[0] = User->getType();
307 bool AllHalfExtracts =
true;
309 for (
User *U :
II.users()) {
311 if (!Ext || !Ext->hasOneUse()) {
312 AllHalfExtracts =
false;
317 if (!Tr || !Tr->getType()->isHalfTy()) {
318 AllHalfExtracts =
false;
325 if (!ExtractTruncPairs.
empty() && AllHalfExtracts) {
336 OverloadTys[0] = HalfVecTy;
339 M, ImageDimIntr->
Intr, OverloadTys);
341 II.mutateType(HalfVecTy);
342 II.setCalledFunction(HalfDecl);
345 for (
auto &[Ext, Tr] : ExtractTruncPairs) {
346 Value *Idx = Ext->getIndexOperand();
348 Builder.SetInsertPoint(Tr);
350 Value *HalfExtract = Builder.CreateExtractElement(&
II, Idx);
353 Tr->replaceAllUsesWith(HalfExtract);
356 for (
auto &[Ext, Tr] : ExtractTruncPairs) {
367 if (!ST->hasA16() && !ST->hasG16())
372 bool HasSampler = BaseOpcode->
Sampler;
373 bool FloatCoord =
false;
375 bool OnlyDerivatives =
false;
380 bool AllowI16SExt = !HasSampler;
383 OperandIndex < ImageDimIntr->VAddrEnd; OperandIndex++) {
384 Value *Coord =
II.getOperand(OperandIndex);
387 if (OperandIndex < ImageDimIntr->CoordStart ||
392 OnlyDerivatives =
true;
401 if (!OnlyDerivatives && !ST->hasA16())
402 OnlyDerivatives =
true;
405 if (!OnlyDerivatives && ImageDimIntr->
NumBiasArgs != 0) {
408 "Only image instructions with a sampler can have a bias");
410 OnlyDerivatives =
true;
413 if (OnlyDerivatives && (!ST->hasG16() || ImageDimIntr->
GradientStart ==
421 II,
II,
II.getIntrinsicID(), IC, [&](
auto &Args,
auto &ArgTys) {
422 ArgTys[ImageDimIntr->GradientTyArg] = CoordType;
423 if (!OnlyDerivatives) {
424 ArgTys[ImageDimIntr->CoordTyArg] = CoordType;
427 if (ImageDimIntr->NumBiasArgs != 0)
428 ArgTys[ImageDimIntr->BiasTyArg] = Type::getHalfTy(II.getContext());
434 OperandIndex < EndIndex; OperandIndex++) {
436 convertTo16Bit(*II.getOperand(OperandIndex), IC.Builder);
441 Value *Bias = II.getOperand(ImageDimIntr->BiasIndex);
442 Args[ImageDimIntr->BiasIndex] = convertTo16Bit(*Bias, IC.Builder);
476 if (
I.hasNoSignedZeros() &&
486 Value *Src =
nullptr;
489 if (Src->getType()->isHalfTy())
506 unsigned VWidth = VTy->getNumElements();
509 for (
int i = VWidth - 1; i > 0; --i) {
531 unsigned VWidth = VTy->getNumElements();
537 SVI->getShuffleMask(ShuffleMask);
539 for (
int I = VWidth - 1;
I > 0; --
I) {
540 if (ShuffleMask.empty()) {
591 unsigned LaneArgIdx)
const {
592 unsigned MaskBits = ST->getWavefrontSizeLog2();
599 if (!
Known.isConstant())
606 Value *LaneArg =
II.getArgOperand(LaneArgIdx);
608 ConstantInt::get(LaneArg->
getType(),
Known.getConstant() & DemandedMask);
609 if (MaskedConst != LaneArg) {
610 II.getOperandUse(LaneArgIdx).set(MaskedConst);
622 CallInst *NewCall =
B.CreateCall(&NewCallee,
Ops, OpBundles);
638 if (ST.isWave32() &&
match(V, W32Pred))
640 if (ST.isWave64() &&
match(V, W64Pred))
649 const auto IID =
II.getIntrinsicID();
650 assert(IID == Intrinsic::amdgcn_readlane ||
651 IID == Intrinsic::amdgcn_readfirstlane ||
652 IID == Intrinsic::amdgcn_permlane64);
662 const bool IsReadLane = (IID == Intrinsic::amdgcn_readlane);
666 Value *LaneID =
nullptr;
668 LaneID =
II.getOperand(1);
682 const auto DoIt = [&](
unsigned OpIdx,
686 Ops.push_back(LaneID);
702 return DoIt(0,
II.getCalledFunction());
706 Type *SrcTy = Src->getType();
712 return DoIt(0, Remangled);
720 return DoIt(1,
II.getCalledFunction());
722 return DoIt(0,
II.getCalledFunction());
733 unsigned Depth = 0) {
743 return CI->getZExtValue();
752 std::optional<unsigned>
LHS =
756 std::optional<unsigned>
RHS =
765 return CI ? std::optional<unsigned>(CI->getZExtValue()) : std::nullopt;
773 unsigned WaveSize = ST.getWavefrontSize();
775 for (
unsigned Lane :
seq(WaveSize)) {
777 if (!Val || *Val >= WaveSize)
786template <
unsigned Period>
788 static_assert(
isPowerOf2_32(Period),
"Period must be a power of two");
789 for (
unsigned I = Period,
E = Ids.
size();
I <
E; ++
I)
790 if (Ids[
I] != Ids[
I % Period] + (
I & ~(Period - 1)))
798 for (
unsigned I = 0;
I <
N; ++
I)
814 return Ids[3] << 6 | Ids[2] << 4 | Ids[1] << 2 | Ids[0];
821 for (
unsigned J = 0; J <
N; ++J)
822 if (Ids[J] != (
N - 1) - J)
834 for (
unsigned J = 1; J < 16; ++J)
835 if (Ids[J] != (Ids[0] + J) % 16)
853 unsigned Mask = Ids[0];
856 for (
unsigned J = 0; J < 16; ++J)
857 if (Ids[J] != (Mask ^ J))
867 unsigned Selector = 0;
868 for (
unsigned J = 0; J < 8; ++J)
869 Selector |= Ids[J] << (J * 3);
878 for (
unsigned J = 0; J < 16; ++J)
879 Sel |=
static_cast<uint64_t>(Ids[J] & 0xF) << (J * 4);
886 if (Ids.
size() != 64)
888 for (
unsigned J = 0; J < 64; ++J)
889 if (Ids[J] != (J ^ 32))
900 for (
unsigned J = 0; J < 16; ++J) {
901 if (Ids[J] < 16 || Ids[J] >= 32)
903 if (Ids[J + 16] != Ids[J] - 16)
914static std::optional<unsigned>
923 unsigned AndMask = 0, OrMask = 0, XorMask = 0;
924 for (
unsigned B = 0;
B < 5; ++
B) {
925 unsigned Bit0 = (Ids[0] >>
B) & 1;
926 unsigned Bit1 = (Ids[1u <<
B] >>
B) & 1;
929 XorMask |= Bit0 <<
B;
937 for (
unsigned I :
seq(32u)) {
938 unsigned Expected = ((
I & AndMask) | OrMask) ^ XorMask;
953static std::optional<unsigned>
964 for (
unsigned I = 0;
I < 32; ++
I)
965 if (Ids[
I] != (
I +
N) % 32)
977 return B.CreateIntrinsic(Intrinsic::amdgcn_update_dpp, {Ty},
979 B.getInt32(0xF),
B.getInt32(0xF),
B.getTrue()});
984 return B.CreateIntrinsic(Intrinsic::amdgcn_mov_dpp8, {Val->
getType()},
985 {Val,
B.getInt32(Selector)});
992 return B.CreateIntrinsic(Intrinsic::amdgcn_permlane16, {Ty},
994 B.getInt32(
Hi),
B.getFalse(),
B.getFalse()});
1002 return B.CreateIntrinsic(Intrinsic::amdgcn_permlanex16, {Ty},
1004 B.getInt32(
Hi),
B.getFalse(),
B.getFalse()});
1012 assert(
DL.getTypeSizeInBits(OrigTy) == 32 &&
1013 "ds_swizzle only supports 32-bit operands");
1017 Src =
B.CreatePtrToInt(Src, I32Ty);
1018 else if (OrigTy != I32Ty)
1019 Src =
B.CreateBitCast(Src, I32Ty);
1020 Value *Result =
B.CreateIntrinsic(Intrinsic::amdgcn_ds_swizzle, {},
1023 return B.CreateIntToPtr(Result, OrigTy);
1024 if (OrigTy != I32Ty)
1025 return B.CreateBitCast(Result, OrigTy);
1031 return B.CreateIntrinsic(Intrinsic::amdgcn_permlane64, {Val->
getType()},
1042 [](
const auto &
E) {
return E.value() ==
E.index(); }))
1066 if (ST.hasDPPRowShare()) {
1071 if (ST.hasDPP() && ST.hasGFX10Insts()) {
1081 if (ST.hasPermlane16Insts()) {
1101 if (ST.hasDsSwizzleRotateMode()) {
1114static std::optional<Instruction *>
1118 if (
DL.getTypeSizeInBits(
II.getType()) != 32)
1119 return std::nullopt;
1121 if (!ST.isWaveSizeKnown())
1122 return std::nullopt;
1124 unsigned WaveSize = ST.getWavefrontSize();
1125 bool IsBpermute =
II.getIntrinsicID() == Intrinsic::amdgcn_ds_bpermute;
1126 Value *Src =
II.getArgOperand(IsBpermute ? 1 : 0);
1127 Value *Index =
II.getArgOperand(IsBpermute ? 0 : 1);
1132 for (
unsigned Lane :
seq(WaveSize)) {
1134 if (!Val || (*Val & 3) || (*Val >> 2) >= WaveSize)
1135 return std::nullopt;
1136 Ids[Lane] = *Val >> 2;
1140 return std::nullopt;
1145 return std::nullopt;
1149std::optional<Instruction *>
1153 case Intrinsic::amdgcn_implicitarg_ptr: {
1154 if (
II.getFunction()->hasFnAttribute(
"amdgpu-no-implicitarg-ptr"))
1156 uint64_t ImplicitArgBytes = ST->getImplicitArgNumBytes(*
II.getFunction());
1159 II.getAttributes().getRetDereferenceableOrNullBytes();
1160 if (CurrentOrNullBytes != 0) {
1163 uint64_t NewBytes = std::max(CurrentOrNullBytes, ImplicitArgBytes);
1166 II.removeRetAttr(Attribute::DereferenceableOrNull);
1170 uint64_t CurrentBytes =
II.getAttributes().getRetDereferenceableBytes();
1171 uint64_t NewBytes = std::max(CurrentBytes, ImplicitArgBytes);
1172 if (NewBytes != CurrentBytes) {
1178 return std::nullopt;
1180 case Intrinsic::amdgcn_rcp: {
1181 Value *Src =
II.getArgOperand(0);
1192 if (
II.isStrictFP())
1210 auto IID = SrcCI->getIntrinsicID();
1215 if (IID == Intrinsic::amdgcn_sqrt || IID == Intrinsic::sqrt) {
1225 SrcCI->getModule(), Intrinsic::amdgcn_rsq, {SrcCI->getType()});
1228 II.setFastMathFlags(InnerFMF);
1230 II.setCalledFunction(NewDecl);
1236 case Intrinsic::amdgcn_sqrt:
1237 case Intrinsic::amdgcn_rsq:
1238 case Intrinsic::amdgcn_tanh: {
1239 Value *Src =
II.getArgOperand(0);
1251 if (IID == Intrinsic::amdgcn_sqrt && Src->getType()->isHalfTy()) {
1253 II.getModule(), Intrinsic::sqrt, {II.getType()});
1254 II.setCalledFunction(NewDecl);
1260 case Intrinsic::amdgcn_log:
1261 case Intrinsic::amdgcn_exp2: {
1262 const bool IsLog = IID == Intrinsic::amdgcn_log;
1263 const bool IsExp = IID == Intrinsic::amdgcn_exp2;
1264 Value *Src =
II.getArgOperand(0);
1274 if (
C->isInfinity()) {
1277 if (!
C->isNegative())
1281 if (IsExp &&
C->isNegative())
1285 if (
II.isStrictFP())
1289 Constant *Quieted = ConstantFP::get(Ty,
C->getValue().makeQuiet());
1294 if (
C->isZero() || (
C->getValue().isDenormal() && Ty->isFloatTy())) {
1296 : ConstantFP::get(Ty, 1.0);
1300 if (IsLog &&
C->isNegative())
1308 case Intrinsic::amdgcn_frexp_mant:
1309 case Intrinsic::amdgcn_frexp_exp: {
1310 Value *Src =
II.getArgOperand(0);
1316 if (IID == Intrinsic::amdgcn_frexp_mant) {
1318 II, ConstantFP::get(
II.getContext(), Significand));
1338 case Intrinsic::amdgcn_class: {
1339 Value *Src0 =
II.getArgOperand(0);
1340 Value *Src1 =
II.getArgOperand(1);
1344 II.getModule(), Intrinsic::is_fpclass, Src0->
getType()));
1347 II.setArgOperand(1, ConstantInt::get(Src1->
getType(),
1368 case Intrinsic::amdgcn_cvt_pkrtz: {
1369 auto foldFPTruncToF16RTZ = [](
Value *Arg) ->
Value * {
1382 return ConstantFP::get(HalfTy, Val);
1385 Value *Src =
nullptr;
1387 if (Src->getType()->isHalfTy())
1394 if (
Value *Src0 = foldFPTruncToF16RTZ(
II.getArgOperand(0))) {
1395 if (
Value *Src1 = foldFPTruncToF16RTZ(
II.getArgOperand(1))) {
1405 case Intrinsic::amdgcn_cvt_pknorm_i16:
1406 case Intrinsic::amdgcn_cvt_pknorm_u16:
1407 case Intrinsic::amdgcn_cvt_pk_i16:
1408 case Intrinsic::amdgcn_cvt_pk_u16: {
1409 Value *Src0 =
II.getArgOperand(0);
1410 Value *Src1 =
II.getArgOperand(1);
1422 case Intrinsic::amdgcn_cvt_off_f32_i4: {
1423 Value* Arg =
II.getArgOperand(0);
1437 constexpr size_t ResValsSize = 16;
1438 static constexpr float ResVals[ResValsSize] = {
1439 0.0, 0.0625, 0.125, 0.1875, 0.25, 0.3125, 0.375, 0.4375,
1440 -0.5, -0.4375, -0.375, -0.3125, -0.25, -0.1875, -0.125, -0.0625};
1442 ConstantFP::get(Ty, ResVals[CArg->
getZExtValue() & (ResValsSize - 1)]);
1445 case Intrinsic::amdgcn_ubfe:
1446 case Intrinsic::amdgcn_sbfe: {
1448 Value *Src =
II.getArgOperand(0);
1455 unsigned IntSize = Ty->getIntegerBitWidth();
1460 if ((Width & (IntSize - 1)) == 0) {
1465 if (Width >= IntSize) {
1467 II, 2, ConstantInt::get(CWidth->
getType(), Width & (IntSize - 1)));
1478 ConstantInt::get(COffset->
getType(),
Offset & (IntSize - 1)));
1482 bool Signed = IID == Intrinsic::amdgcn_sbfe;
1484 if (!CWidth || !COffset)
1494 if (
Offset + Width < IntSize) {
1498 RightShift->takeName(&
II);
1505 RightShift->takeName(&
II);
1508 case Intrinsic::amdgcn_exp:
1509 case Intrinsic::amdgcn_exp_row:
1510 case Intrinsic::amdgcn_exp_compr: {
1516 bool IsCompr = IID == Intrinsic::amdgcn_exp_compr;
1518 for (
int I = 0;
I < (IsCompr ? 2 : 4); ++
I) {
1519 if ((!IsCompr && (EnBits & (1 <<
I)) == 0) ||
1520 (IsCompr && ((EnBits & (0x3 << (2 *
I))) == 0))) {
1521 Value *Src =
II.getArgOperand(
I + 2);
1535 case Intrinsic::amdgcn_fmed3: {
1536 Value *Src0 =
II.getArgOperand(0);
1537 Value *Src1 =
II.getArgOperand(1);
1538 Value *Src2 =
II.getArgOperand(2);
1540 for (
Value *Src : {Src0, Src1, Src2}) {
1545 if (
II.isStrictFP())
1582 const APFloat *ConstSrc0 =
nullptr;
1583 const APFloat *ConstSrc1 =
nullptr;
1584 const APFloat *ConstSrc2 =
nullptr;
1589 const bool IsPosInfinity = ConstSrc0 && ConstSrc0->
isPosInfinity();
1609 const bool IsPosInfinity = ConstSrc1 && ConstSrc1->
isPosInfinity();
1632 auto *Quieted = ConstantFP::get(
II.getType(), ConstSrc2->
makeQuiet());
1652 CI->copyFastMathFlags(&
II);
1678 II.setArgOperand(0, Src0);
1679 II.setArgOperand(1, Src1);
1680 II.setArgOperand(2, Src2);
1690 ConstantFP::get(
II.getType(), Result));
1695 if (!ST->hasMed3_16())
1704 IID, {
X->getType()}, {
X,
Y, Z}, &
II,
II.getName());
1712 case Intrinsic::amdgcn_icmp:
1713 case Intrinsic::amdgcn_fcmp: {
1717 bool IsInteger = IID == Intrinsic::amdgcn_icmp;
1724 Value *Src0 =
II.getArgOperand(0);
1725 Value *Src1 =
II.getArgOperand(1);
1752 II.setArgOperand(0, Src1);
1753 II.setArgOperand(1, Src0);
1755 2, ConstantInt::get(CC->
getType(),
static_cast<int>(SwapPred)));
1802 ? Intrinsic::amdgcn_fcmp
1803 : Intrinsic::amdgcn_icmp;
1808 unsigned Width = CmpType->getBitWidth();
1809 unsigned NewWidth = Width;
1817 else if (Width <= 32)
1819 else if (Width <= 64)
1824 if (Width != NewWidth) {
1834 }
else if (!Ty->isFloatTy() && !Ty->isDoubleTy() && !Ty->isHalfTy())
1837 Value *Args[] = {SrcLHS, SrcRHS,
1838 ConstantInt::get(CC->
getType(), SrcPred)};
1840 NewIID, {
II.getType(), SrcLHS->
getType()}, Args);
1847 case Intrinsic::amdgcn_mbcnt_hi:
1852 case Intrinsic::amdgcn_mbcnt_lo: {
1865 if (std::optional<ConstantRange> ExistingRange =
II.getRange()) {
1866 ComputedRange = ComputedRange.
intersectWith(*ExistingRange);
1867 if (ComputedRange == *ExistingRange)
1871 II.addRangeRetAttr(ComputedRange);
1874 case Intrinsic::amdgcn_ballot: {
1875 Value *Arg =
II.getArgOperand(0);
1880 if (Src->isZero()) {
1885 if (ST->isWave32() &&
II.getType()->getIntegerBitWidth() == 64) {
1892 {IC.Builder.getInt32Ty()},
1893 {II.getArgOperand(0)}),
1900 case Intrinsic::amdgcn_wavefrontsize: {
1901 if (ST->isWaveSizeKnown())
1903 II, ConstantInt::get(
II.getType(), ST->getWavefrontSize()));
1906 case Intrinsic::amdgcn_wqm_vote: {
1913 case Intrinsic::amdgcn_kill: {
1915 if (!
C || !
C->getZExtValue())
1921 case Intrinsic::amdgcn_s_sendmsg:
1922 case Intrinsic::amdgcn_s_sendmsghalt: {
1928 Value *M0Val =
II.getArgOperand(1);
1934 decodeMsg(MsgImm->getZExtValue(), MsgId, OpId, StreamId, *ST);
1936 if (!msgDoesNotUseM0(MsgId, *ST))
1940 II.dropUBImplyingAttrsAndMetadata();
1944 case Intrinsic::amdgcn_update_dpp: {
1945 Value *Old =
II.getArgOperand(0);
1950 if (BC->isNullValue() || RM->getZExtValue() != 0xF ||
1957 case Intrinsic::amdgcn_permlane16:
1958 case Intrinsic::amdgcn_permlane16_var:
1959 case Intrinsic::amdgcn_permlanex16:
1960 case Intrinsic::amdgcn_permlanex16_var: {
1962 Value *VDstIn =
II.getArgOperand(0);
1967 unsigned int FiIdx = (IID == Intrinsic::amdgcn_permlane16 ||
1968 IID == Intrinsic::amdgcn_permlanex16)
1975 unsigned int BcIdx = FiIdx + 1;
1984 case Intrinsic::amdgcn_wave_shuffle:
1986 case Intrinsic::amdgcn_permlane64:
1987 case Intrinsic::amdgcn_readfirstlane:
1988 case Intrinsic::amdgcn_readlane:
1989 case Intrinsic::amdgcn_ds_bpermute: {
1991 unsigned SrcIdx = IID == Intrinsic::amdgcn_ds_bpermute ? 1 : 0;
1992 const Use &Src =
II.getArgOperandUse(SrcIdx);
1996 if (IID == Intrinsic::amdgcn_readlane &&
2003 if (IID == Intrinsic::amdgcn_ds_bpermute) {
2004 const Use &Lane =
II.getArgOperandUse(0);
2008 II.getModule(), Intrinsic::amdgcn_readlane,
II.getType());
2009 II.setCalledFunction(NewDecl);
2010 II.setOperand(0, Src);
2011 II.setOperand(1, NewLane);
2016 if (IID == Intrinsic::amdgcn_ds_bpermute)
2022 return std::nullopt;
2024 case Intrinsic::amdgcn_writelane: {
2028 return std::nullopt;
2030 case Intrinsic::amdgcn_trig_preop: {
2033 if (!
II.getType()->isDoubleTy())
2036 Value *Src =
II.getArgOperand(0);
2037 Value *Segment =
II.getArgOperand(1);
2046 if (StrippedSign != Src)
2049 if (
II.isStrictFP())
2071 unsigned Shift = SegmentVal * 53;
2076 static const uint32_t TwoByPi[] = {
2077 0xa2f9836e, 0x4e441529, 0xfc2757d1, 0xf534ddc0, 0xdb629599, 0x3c439041,
2078 0xfe5163ab, 0xdebbc561, 0xb7246e3a, 0x424dd2e0, 0x06492eea, 0x09d1921c,
2079 0xfe1deb1c, 0xb129a73e, 0xe88235f5, 0x2ebb4484, 0xe99c7026, 0xb45f7e41,
2080 0x3991d639, 0x835339f4, 0x9c845f8b, 0xbdf9283b, 0x1ff897ff, 0xde05980f,
2081 0xef2f118b, 0x5a0a6d1f, 0x6d367ecf, 0x27cb09b7, 0x4f463f66, 0x9e5fea2d,
2082 0x7527bac7, 0xebe5f17b, 0x3d0739f7, 0x8a5292ea, 0x6bfb5fb1, 0x1f8d5d08,
2086 unsigned Idx = Shift >> 5;
2087 if (Idx + 2 >= std::size(TwoByPi)) {
2092 unsigned BShift = Shift & 0x1f;
2096 Thi = (Thi << BShift) | (Tlo >> (64 - BShift));
2100 int Scale = -53 - Shift;
2107 case Intrinsic::amdgcn_fmul_legacy: {
2108 Value *Op0 =
II.getArgOperand(0);
2109 Value *Op1 =
II.getArgOperand(1);
2111 for (
Value *Src : {Op0, Op1}) {
2132 case Intrinsic::amdgcn_fma_legacy: {
2133 Value *Op0 =
II.getArgOperand(0);
2134 Value *Op1 =
II.getArgOperand(1);
2135 Value *Op2 =
II.getArgOperand(2);
2137 for (
Value *Src : {Op0, Op1, Op2}) {
2159 II.getModule(), Intrinsic::fma,
II.getType()));
2164 case Intrinsic::amdgcn_is_shared:
2165 case Intrinsic::amdgcn_is_private: {
2166 Value *Src =
II.getArgOperand(0);
2176 case Intrinsic::amdgcn_make_buffer_rsrc: {
2177 Value *Src =
II.getArgOperand(0);
2180 return std::nullopt;
2182 case Intrinsic::amdgcn_raw_buffer_store_format:
2183 case Intrinsic::amdgcn_struct_buffer_store_format:
2184 case Intrinsic::amdgcn_raw_tbuffer_store:
2185 case Intrinsic::amdgcn_struct_tbuffer_store:
2186 case Intrinsic::amdgcn_image_store_1d:
2187 case Intrinsic::amdgcn_image_store_1darray:
2188 case Intrinsic::amdgcn_image_store_2d:
2189 case Intrinsic::amdgcn_image_store_2darray:
2190 case Intrinsic::amdgcn_image_store_2darraymsaa:
2191 case Intrinsic::amdgcn_image_store_2dmsaa:
2192 case Intrinsic::amdgcn_image_store_3d:
2193 case Intrinsic::amdgcn_image_store_cube:
2194 case Intrinsic::amdgcn_image_store_mip_1d:
2195 case Intrinsic::amdgcn_image_store_mip_1darray:
2196 case Intrinsic::amdgcn_image_store_mip_2d:
2197 case Intrinsic::amdgcn_image_store_mip_2darray:
2198 case Intrinsic::amdgcn_image_store_mip_3d:
2199 case Intrinsic::amdgcn_image_store_mip_cube: {
2204 if (ST->hasDefaultComponentBroadcast())
2206 else if (ST->hasDefaultComponentZero())
2211 int DMaskIdx = getAMDGPUImageDMaskIntrinsic(
II.getIntrinsicID()) ? 1 : -1;
2219 case Intrinsic::amdgcn_prng_b32: {
2220 auto *Src =
II.getArgOperand(0);
2224 return std::nullopt;
2226 case Intrinsic::amdgcn_mfma_scale_f32_16x16x128_f8f6f4:
2227 case Intrinsic::amdgcn_mfma_scale_f32_32x32x64_f8f6f4: {
2228 Value *Src0 =
II.getArgOperand(0);
2229 Value *Src1 =
II.getArgOperand(1);
2235 auto getFormatNumRegs = [](
unsigned FormatVal) {
2236 switch (FormatVal) {
2250 bool MadeChange =
false;
2251 unsigned Src0NumElts = getFormatNumRegs(CBSZ);
2252 unsigned Src1NumElts = getFormatNumRegs(BLGP);
2256 if (Src0Ty->getNumElements() > Src0NumElts) {
2263 if (Src1Ty->getNumElements() > Src1NumElts) {
2271 return std::nullopt;
2282 case Intrinsic::amdgcn_wmma_f32_16x16x128_f8f6f4:
2283 case Intrinsic::amdgcn_wmma_scale_f32_16x16x128_f8f6f4:
2284 case Intrinsic::amdgcn_wmma_scale16_f32_16x16x128_f8f6f4: {
2285 Value *Src0 =
II.getArgOperand(1);
2286 Value *Src1 =
II.getArgOperand(3);
2292 bool MadeChange =
false;
2298 if (Src0Ty->getNumElements() > Src0NumElts) {
2305 if (Src1Ty->getNumElements() > Src1NumElts) {
2313 return std::nullopt;
2330 return std::nullopt;
2343 int DMaskIdx,
bool IsLoad) {
2346 :
II.getOperand(0)->getType());
2347 unsigned VWidth = IIVTy->getNumElements();
2350 Type *EltTy = IIVTy->getElementType();
2362 const unsigned UnusedComponentsAtFront = DemandedElts.
countr_zero();
2367 DemandedElts = (1 << ActiveBits) - 1;
2369 if (UnusedComponentsAtFront > 0) {
2370 static const unsigned InvalidOffsetIdx = 0xf;
2373 switch (
II.getIntrinsicID()) {
2374 case Intrinsic::amdgcn_raw_buffer_load:
2375 case Intrinsic::amdgcn_raw_ptr_buffer_load:
2378 case Intrinsic::amdgcn_s_buffer_load:
2379 case Intrinsic::amdgcn_ptr_s_buffer_load:
2383 if (ActiveBits == 4 && UnusedComponentsAtFront == 1)
2384 OffsetIdx = InvalidOffsetIdx;
2388 case Intrinsic::amdgcn_struct_buffer_load:
2389 case Intrinsic::amdgcn_struct_ptr_buffer_load:
2394 OffsetIdx = InvalidOffsetIdx;
2398 if (OffsetIdx != InvalidOffsetIdx) {
2400 DemandedElts &= ~((1 << UnusedComponentsAtFront) - 1);
2401 auto *
Offset = Args[OffsetIdx];
2402 unsigned SingleComponentSizeInBits =
2404 unsigned OffsetAdd =
2405 UnusedComponentsAtFront * SingleComponentSizeInBits / 8;
2406 auto *OffsetAddVal = ConstantInt::get(
Offset->getType(), OffsetAdd);
2426 unsigned NewDMaskVal = 0;
2427 unsigned OrigLdStIdx = 0;
2428 for (
unsigned SrcIdx = 0; SrcIdx < 4; ++SrcIdx) {
2429 const unsigned Bit = 1 << SrcIdx;
2430 if (!!(DMaskVal & Bit)) {
2431 if (!!DemandedElts[OrigLdStIdx])
2437 if (DMaskVal != NewDMaskVal)
2438 Args[DMaskIdx] = ConstantInt::get(DMask->
getType(), NewDMaskVal);
2441 unsigned NewNumElts = DemandedElts.
popcount();
2445 if (NewNumElts >= VWidth && DemandedElts.
isMask()) {
2447 II.setArgOperand(DMaskIdx, Args[DMaskIdx]);
2459 OverloadTys[0] = NewTy;
2463 for (
unsigned OrigStoreIdx = 0; OrigStoreIdx < VWidth; ++OrigStoreIdx)
2464 if (DemandedElts[OrigStoreIdx])
2467 if (NewNumElts == 1)
2474 II.getIntrinsicID(), OverloadTys, Args);
2477 AttributeList OldAttrList =
II.getAttributes();
2481 if (NewNumElts == 1) {
2487 unsigned NewLoadIdx = 0;
2488 for (
unsigned OrigLoadIdx = 0; OrigLoadIdx < VWidth; ++OrigLoadIdx) {
2489 if (!!DemandedElts[OrigLoadIdx])
2505 APInt &UndefElts)
const {
2510 const unsigned FirstElt = DemandedElts.
countr_zero();
2512 const unsigned MaskLen = LastElt - FirstElt + 1;
2514 unsigned OldNumElts = VT->getNumElements();
2515 if (MaskLen == OldNumElts && MaskLen != 1)
2518 Type *EltTy = VT->getElementType();
2526 Value *Src =
II.getArgOperand(0);
2531 II.getOperandBundlesAsDefs(OpBundles);
2548 for (
unsigned I = 0;
I != MaskLen; ++
I) {
2549 if (DemandedElts[FirstElt +
I])
2550 ExtractMask[
I] = FirstElt +
I;
2559 for (
unsigned I = 0;
I != MaskLen; ++
I) {
2560 if (DemandedElts[FirstElt +
I])
2561 InsertMask[FirstElt +
I] =
I;
2573 SimplifyAndSetOp)
const {
2574 switch (
II.getIntrinsicID()) {
2575 case Intrinsic::amdgcn_readfirstlane:
2576 SimplifyAndSetOp(&
II, 0, DemandedElts, UndefElts);
2578 case Intrinsic::amdgcn_raw_buffer_load:
2579 case Intrinsic::amdgcn_raw_ptr_buffer_load:
2580 case Intrinsic::amdgcn_raw_buffer_load_format:
2581 case Intrinsic::amdgcn_raw_ptr_buffer_load_format:
2582 case Intrinsic::amdgcn_raw_tbuffer_load:
2583 case Intrinsic::amdgcn_raw_ptr_tbuffer_load:
2584 case Intrinsic::amdgcn_s_buffer_load:
2585 case Intrinsic::amdgcn_ptr_s_buffer_load:
2586 case Intrinsic::amdgcn_struct_buffer_load:
2587 case Intrinsic::amdgcn_struct_ptr_buffer_load:
2588 case Intrinsic::amdgcn_struct_buffer_load_format:
2589 case Intrinsic::amdgcn_struct_ptr_buffer_load_format:
2590 case Intrinsic::amdgcn_struct_tbuffer_load:
2591 case Intrinsic::amdgcn_struct_ptr_tbuffer_load:
2594 if (getAMDGPUImageDMaskIntrinsic(
II.getIntrinsicID())) {
2600 return std::nullopt;
for(const MachineOperand &MO :llvm::drop_begin(OldMI.operands(), Desc.getNumOperands()))
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static Value * createPermlane16(IRBuilderBase &B, Value *Val, uint32_t Lo, uint32_t Hi)
Emit v_permlane16 with the precomputed lane-select halves.
static std::optional< unsigned > matchRowSharePattern(ArrayRef< uint8_t > Ids)
Match a row-share pattern: all 16 lanes of each row read the same source lane.
static bool matchMirrorPattern(ArrayRef< uint8_t > Ids)
Match an N-lane reversal (mirror) pattern.
static bool canSafelyConvertTo16Bit(Value &V, bool IsFloat, bool AllowI16SExt=false)
static bool tryBuildShuffleMap(Value *Index, const GCNSubtarget &ST, SmallVectorImpl< uint8_t > &Ids, const DataLayout &DL)
Build the per-lane shuffle map by evaluating Index for every lane in the wave.
static std::optional< unsigned > matchQuadPermPattern(ArrayRef< uint8_t > Ids)
Match a 4-lane (quad) permutation, encoded as the v_mov_b32_dpp QUAD_PERM control word: bits[1:0]=Ids...
static std::optional< unsigned > matchDsSwizzleRotatePattern(ArrayRef< uint8_t > Ids)
Match a GFX9+ DS_SWIZZLE rotate-mode permutation: a cyclic left-rotation of all 32 lanes within each ...
static std::optional< unsigned > matchHalfRowPermPattern(ArrayRef< uint8_t > Ids)
Match an 8-lane arbitrary permutation, encoded as the v_mov_b32_dpp8 24-bit selector (three bits per ...
static std::optional< unsigned > matchRowXMaskPattern(ArrayRef< uint8_t > Ids)
Match an XOR mask pattern within each 16-lane row: Ids[J] == Mask ^ J, with Mask in [1,...
static constexpr auto matchHalfRowMirrorPattern
static Value * createPermlaneX16(IRBuilderBase &B, Value *Val, uint32_t Lo, uint32_t Hi)
Emit v_permlanex16 with the precomputed lane-select halves.
static bool isRowPattern(ArrayRef< uint8_t > Ids)
Match an N-lane row pattern: each lane in [0, N) reads from a source lane in the same N-lane row,...
static bool canContractSqrtToRsq(const FPMathOperator *SqrtOp)
Return true if it's legal to contract llvm.amdgcn.rcp(llvm.sqrt)
static bool isTriviallyUniform(const Use &U)
Return true if we can easily prove that use U is uniform.
static CallInst * rewriteCall(IRBuilderBase &B, CallInst &Old, Function &NewCallee, ArrayRef< Value * > Ops)
static Value * convertTo16Bit(Value &V, InstCombiner::BuilderTy &Builder)
static constexpr auto isFullRowPattern
static constexpr auto isQuadPattern
static APInt trimTrailingZerosInVector(InstCombiner &IC, Value *UseV, Instruction *I)
static uint64_t computePermlane16Masks(ArrayRef< uint8_t > Ids)
Pack a 16-lane permutation into a single 64-bit value: four bits per output lane, lane J in bits [J*4...
static bool matchHalfWaveSwapPattern(ArrayRef< uint8_t > Ids)
Match a half-wave swap: lane J reads from lane J ^ 32.
static bool hasPeriodicLayout(ArrayRef< uint8_t > Ids)
Lanes are partitioned into groups of Period; each group is a translated copy of the first: Ids[I] = I...
static std::optional< Instruction * > tryOptimizeShufflePattern(InstCombiner &IC, IntrinsicInst &II, const GCNSubtarget &ST)
Try to fold a wave_shuffle/ds_bpermute whose lane index is a constant function of the lane ID into a ...
static constexpr auto isHalfRowPattern
static APInt defaultComponentBroadcast(Value *V)
static std::optional< unsigned > matchDsSwizzleBitmaskPattern(ArrayRef< uint8_t > Ids)
Match a DS_SWIZZLE bitmask-mode permutation: dst_lane = ((src_lane & AND) | OR) ^ XOR with each mask ...
static Value * createDsSwizzle(IRBuilderBase &B, Value *Val, unsigned Offset, const DataLayout &DL)
Emit ds_swizzle with the given immediate, bitcasting/converting between pointer/float types and i32 a...
static std::optional< Instruction * > modifyIntrinsicCall(IntrinsicInst &OldIntr, Instruction &InstToReplace, unsigned NewIntr, InstCombiner &IC, std::function< void(SmallVectorImpl< Value * > &, SmallVectorImpl< Type * > &)> Func)
Applies Func(OldIntr.Args, OldIntr.ArgTys), creates intrinsic call with modified arguments (based on ...
static Value * matchShuffleToHWIntrinsic(IRBuilderBase &B, Value *Src, ArrayRef< uint8_t > Ids, const GCNSubtarget &ST, const DataLayout &DL)
Given a shuffle map, try to emit the best hardware intrinsic.
static std::optional< unsigned > matchRowRotatePattern(ArrayRef< uint8_t > Ids)
Match a 16-lane cyclic rotation; returns the rotation amount in [1, 15].
static bool isCrossRowPattern(ArrayRef< uint8_t > Ids)
Match a cross-row permutation suitable for v_permlanex16: every lane in the low 16-lane half reads fr...
static bool isThreadID(const GCNSubtarget &ST, Value *V)
static Value * createUpdateDpp(IRBuilderBase &B, Value *Val, unsigned Ctrl)
Emit v_mov_b32_dpp with the given control word, row/bank masks 0xF, and bound_ctrl=1 so out-of-bounds...
static APFloat fmed3AMDGCN(const APFloat &Src0, const APFloat &Src1, const APFloat &Src2)
static Value * simplifyAMDGCNMemoryIntrinsicDemanded(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, int DMaskIdx=-1, bool IsLoad=true)
Implement SimplifyDemandedVectorElts for amdgcn buffer and image intrinsics.
static std::optional< Instruction * > simplifyAMDGCNImageIntrinsic(const GCNSubtarget *ST, const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr, IntrinsicInst &II, InstCombiner &IC)
static Value * createMovDpp8(IRBuilderBase &B, Value *Val, unsigned Selector)
Emit v_mov_b32_dpp8 with the given 24-bit lane selector.
static Value * matchFPExtFromF16(Value *Arg)
Match an fpext from half to float, or a constant we can convert.
static constexpr auto matchFullRowMirrorPattern
static std::optional< unsigned > evalLaneExpr(Value *V, unsigned Lane, const GCNSubtarget &ST, const DataLayout &DL, unsigned Depth=0)
Evaluate V as a function of the lane ID and return its value on Lane, or std::nullopt if V is not a c...
static Value * createPermlane64(IRBuilderBase &B, Value *Val)
Emit v_permlane64 (swap of the two 32-lane halves of a wave64).
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
Utilities for dealing with flags related to floating point properties and mode controls.
AMD GCN specific subclass of TargetSubtarget.
This file provides the interface for the instcombine pass implementation.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
uint64_t IntrinsicInst * II
Provides some synthesis utilities to produce sequences of values.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static const fltSemantics & IEEEsingle()
static constexpr roundingMode rmTowardZero
static constexpr roundingMode rmNearestTiesToEven
static const fltSemantics & IEEEhalf()
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
bool bitwiseIsEqual(const APFloat &RHS) const
bool isPosInfinity() const
APFloat makeQuiet() const
Assuming this is an IEEE-754 NaN value, quiet its signaling bit.
APInt bitcastToAPInt() const
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Class for arbitrary precision integers.
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
void clearBit(unsigned BitPosition)
Set a given bit to 0.
uint64_t getZExtValue() const
Get zero extended value.
unsigned popcount() const
Count the number of bits set.
LLVM_ABI uint64_t extractBitsAsZExtValue(unsigned numBits, unsigned bitPosition) const
unsigned getActiveBits() const
Compute the number of active bits in the value.
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
unsigned countr_zero() const
Count the number of trailing zero bits.
bool isMask(unsigned numBits) const
Represent a constant reference to an array (0 or more elements consecutively in memory),...
ArrayRef< T > take_front(size_t N=1) const
Return a copy of *this with only the first N elements.
size_t size() const
Get the array size.
static LLVM_ABI Attribute getWithDereferenceableBytes(LLVMContext &Context, uint64_t Bytes)
LLVM_ABI const Module * getModule() const
Return the module owning the function this basic block belongs to, or nullptr if the function does no...
bool isTypeLegal(Type *Ty) const override
LLVM_ABI void getOperandBundlesAsDefs(SmallVectorImpl< OperandBundleDef > &Defs) const
Return the list of operand bundles attached to this instruction as a vector of OperandBundleDefs.
Function * getCalledFunction() const
Returns the function called, or null if this is an indirect function invocation or the function signa...
void setAttributes(AttributeList A)
Set the attributes for this call.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
AttributeList getAttributes() const
Return the attributes for this call.
This class represents a function call, abstracting a target machine's calling convention.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
bool isFPPredicate() const
Predicate getInversePredicate() const
For example, EQ -> NE, UGT -> ULE, SLT -> SGE, OEQ -> UNE, UGT -> OLE, OLT -> UGE,...
An abstraction over a floating-point predicate, and a pack of an integer predicate with samesign info...
ConstantFP - Floating Point Values [float, double].
const APFloat & getValueAPF() const
static LLVM_ABI ConstantFP * getZero(Type *Ty, bool Negative=false)
static LLVM_ABI ConstantFP * getNaN(Type *Ty, bool Negative=false, uint64_t Payload=0)
static LLVM_ABI ConstantFP * getInfinity(Type *Ty, bool Negative=false)
This is the shared class of boolean and integer constants.
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
static LLVM_ABI ConstantInt * getFalse(LLVMContext &Context)
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
const APInt & getValue() const
Return the constant as an APInt value reference.
This class represents a range of values.
LLVM_ABI ConstantRange add(const ConstantRange &Other) const
Return a new range representing the possible values resulting from an addition of a value in this ran...
LLVM_ABI bool isFullSet() const
Return true if this set contains all of the elements possible for this data-type.
LLVM_ABI ConstantRange intersectWith(const ConstantRange &CR, PreferredRangeType Type=Smallest) const
Return the range that results from the intersection of this range with another range.
This is an important base class in LLVM.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
A parsed version of the target data layout string in and methods for querying it.
TypeSize getTypeSizeInBits(Type *Ty) const
Size examples:
LLVM_ABI bool dominates(const BasicBlock *BB, const Use &U) const
Return true if the (end of the) basic block BB dominates the use U.
Tagged union holding either a T or a Error.
This class represents an extension of floating point types.
Utility class for floating point operations which can have information about relaxed accuracy require...
FastMathFlags getFastMathFlags() const
Convenience function for getting all the fast-math flags.
bool hasApproxFunc() const
Test if this operation allows approximations of math library functions or intrinsics.
LLVM_ABI float getFPAccuracy() const
Get the maximum error permitted by this operation in ULPs.
Convenience struct for specifying and reasoning about fast-math flags.
bool allowContract() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
bool simplifyDemandedLaneMaskArg(InstCombiner &IC, IntrinsicInst &II, unsigned LaneAgIdx) const
Simplify a lane index operand (e.g.
std::optional< Instruction * > instCombineIntrinsic(InstCombiner &IC, IntrinsicInst &II) const override
Instruction * hoistLaneIntrinsicThroughOperand(InstCombiner &IC, IntrinsicInst &II) const
std::optional< Value * > simplifyDemandedVectorEltsIntrinsic(InstCombiner &IC, IntrinsicInst &II, APInt DemandedElts, APInt &UndefElts, APInt &UndefElts2, APInt &UndefElts3, std::function< void(Instruction *, unsigned, APInt, APInt &)> SimplifyAndSetOp) const override
KnownIEEEMode fpenvIEEEMode(const Instruction &I) const
Return KnownIEEEMode::On if we know if the use context can assume "amdgpu-ieee"="true" and KnownIEEEM...
Value * simplifyAMDGCNLaneIntrinsicDemanded(InstCombiner &IC, IntrinsicInst &II, const APInt &DemandedElts, APInt &UndefElts) const
bool canSimplifyLegacyMulToMul(const Instruction &I, const Value *Op0, const Value *Op1, InstCombiner &IC) const
Common base class shared among various IRBuilders.
LLVM_ABI CallInst * CreateIntrinsicWithoutFolding(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={})
Create a call to intrinsic ID with Args, mangled using OverloadTypes.
Value * CreateInsertElement(Type *VecTy, Value *NewElt, Value *Idx, const Twine &Name="")
Value * CreateExtractElement(Value *Vec, Value *Idx, const Twine &Name="")
IntegerType * getIntNTy(unsigned N)
Fetch the type representing an N-bit integer.
Value * CreateZExtOrTrunc(Value *V, Type *DestTy, const Twine &Name="")
Create a ZExt or Trunc from the integer value V to DestTy.
ConstantInt * getTrue()
Get the constant value for i1 true.
Value * CreateSExt(Value *V, Type *DestTy, const Twine &Name="")
Value * CreateLShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateExtractVector(Type *DstType, Value *SrcVec, Value *Idx, const Twine &Name="")
Create a call to the vector.extract intrinsic.
BasicBlock * GetInsertBlock() const
Value * CreateICmpNE(Value *LHS, Value *RHS, const Twine &Name="")
ConstantInt * getInt64(uint64_t C)
Get a constant 64-bit value.
Value * CreateMaxNum(Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create call to the maxnum intrinsic.
Value * CreateShl(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
Value * CreateZExt(Value *V, Type *DestTy, const Twine &Name="", bool IsNonNeg=false)
Value * CreateShuffleVector(Value *V1, Value *V2, Value *Mask, const Twine &Name="")
LLVM_ABI Value * CreateIntrinsic(Intrinsic::ID ID, ArrayRef< Type * > OverloadTypes, ArrayRef< Value * > Args, FMFSource FMFSource={}, const Twine &Name="", ArrayRef< OperandBundleDef > OpBundles={}, function_ref< void(CallInst *)> SetFn=[](CallInst *) {})
Variant to create a possibly constant-folded intrinsic.
Value * CreateMaximumNum(Value *LHS, Value *RHS, const Twine &Name="")
Create call to the maximum intrinsic.
Value * CreateMinNum(Value *LHS, Value *RHS, FMFSource FMFSource={}, const Twine &Name="")
Create call to the minnum intrinsic.
Value * CreateAdd(Value *LHS, Value *RHS, const Twine &Name="", bool HasNUW=false, bool HasNSW=false)
CallInst * CreateCall(FunctionType *FTy, Value *Callee, ArrayRef< Value * > Args={}, const Twine &Name="", MDNode *FPMathTag=nullptr)
void SetInsertPoint(BasicBlock *TheBB)
This specifies that created instructions should be appended to the end of the specified block.
Value * CreateFAddFMF(Value *L, Value *R, FMFSource FMFSource, const Twine &Name="", MDNode *FPMD=nullptr)
Value * CreateMinimumNum(Value *LHS, Value *RHS, const Twine &Name="")
Create call to the minimumnum intrinsic.
Value * CreateAShr(Value *LHS, Value *RHS, const Twine &Name="", bool isExact=false)
Value * CreateFMulFMF(Value *L, Value *R, FMFSource FMFSource, const Twine &Name="", MDNode *FPMD=nullptr)
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
The core instruction combiner logic.
const DataLayout & getDataLayout() const
virtual Instruction * eraseInstFromFunction(Instruction &I)=0
Combiner aware instruction erasure.
DominatorTree & getDominatorTree() const
Instruction * replaceInstUsesWith(Instruction &I, Value *V)
A combiner-aware RAUW-like routine.
virtual bool SimplifyDemandedBits(Instruction *I, unsigned OpNo, const APInt &DemandedMask, KnownBits &Known, const SimplifyQuery &Q, unsigned Depth=0)=0
IRBuilder< TargetFolder, IRBuilderInstCombineInserter > BuilderTy
An IRBuilder that automatically inserts new instructions into the worklist.
static Value * stripSignOnlyFPOps(Value *Val)
Ignore all operations which only change the sign of a value, returning the underlying magnitude value...
Instruction * replaceOperand(Instruction &I, unsigned OpNum, Value *V)
Replace operand of instruction and add old operand to the worklist.
const SimplifyQuery & getSimplifyQuery() const
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void copyFastMathFlags(FastMathFlags FMF)
Convenience function for transferring all fast-math flag values to this instruction,...
LLVM_ABI void copyMetadata(const Instruction &SrcInst, ArrayRef< unsigned > WL=ArrayRef< unsigned >())
Copy metadata from SrcInst to this instruction.
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
A Module instance is used to store all the information related to an LLVM module.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
The instances of the Type class are immutable: once they are created, they are never changed.
bool isPointerTy() const
True if this is an instance of PointerType.
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
LLVM_ABI Type * getWithNewType(Type *EltTy) const
Given vector type, change the element type, whilst keeping the old number of elements.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI Type * getHalfTy(LLVMContext &C)
bool isVoidTy() const
Return true if this is 'void'.
static LLVM_ABI UndefValue * get(Type *T)
Static factory methods - Return an 'undef' object of the specified type.
A Use represents the edge between a Value definition and its users.
const Use & getOperandUse(unsigned i) const
void setOperand(unsigned i, Value *Val)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI bool hasOneUser() const
Return true if there is exactly one user of this value.
LLVMContext & getContext() const
All values hold a context through their type.
LLVM_ABI void takeName(Value *V)
Transfer the name from V to this value.
const ParentTy * getParent() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
LLVM_READONLY const MIMGOffsetMappingInfo * getMIMGOffsetMappingInfo(unsigned Offset)
uint8_t wmmaScaleF8F6F4FormatToNumRegs(unsigned Fmt)
const ImageDimIntrinsicInfo * getImageDimIntrinsicByBaseOpcode(unsigned BaseOpcode, unsigned Dim)
LLVM_READONLY const MIMGMIPMappingInfo * getMIMGMIPMappingInfo(unsigned MIP)
bool isArgPassedInSGPR(const Argument *A)
bool isIntrinsicAlwaysUniform(unsigned IntrID)
LLVM_READONLY const MIMGBiasMappingInfo * getMIMGBiasMappingInfo(unsigned Bias)
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
LLVM_READONLY const MIMGLZMappingInfo * getMIMGLZMappingInfo(unsigned L)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfo(unsigned DimEnum)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI bool isSignatureValid(Intrinsic::ID ID, FunctionType *FT, SmallVectorImpl< Type * > &OverloadTys, raw_ostream &OS=nulls())
Returns true if FT is a valid function type for intrinsic ID.
OneUse_match< SubPat > m_OneUse(const SubPat &SP)
cst_pred_ty< is_all_ones > m_AllOnes()
Match an integer or vector with all bits set.
auto m_Cmp()
Matches any compare instruction and ignore it.
bool match(Val *V, const Pattern &P)
match_bind< Instruction > m_Instruction(Instruction *&I)
Match an instruction, capturing it if we match.
cstfp_pred_ty< is_any_zero_fp > m_AnyZeroFP()
Match a floating-point negative zero or positive zero.
ap_match< APFloat > m_APFloat(const APFloat *&Res)
Match a ConstantFP or splatted ConstantVector, binding the specified pointer to the contained APFloat...
TwoOps_match< Val_t, Idx_t, Instruction::ExtractElement > m_ExtractElt(const Val_t &Val, const Idx_t &Idx)
Matches ExtractElementInst.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
auto m_Value()
Match an arbitrary value and ignore it.
CastInst_match< OpTy, FPExtInst > m_FPExt(const OpTy &Op)
CastInst_match< OpTy, ZExtInst > m_ZExt(const OpTy &Op)
Matches ZExt.
auto m_Intrinsic(const Ts &...Ops)
Match intrinsic calls like this: m_Intrinsic<Intrinsic::fabs>(m_Value(X))
match_combine_or< CastInst_match< OpTy, ZExtInst >, CastInst_match< OpTy, SExtInst > > m_ZExtOrSExt(const OpTy &Op)
auto m_ConstantFP()
Match an arbitrary ConstantFP and ignore it.
CastInst_match< OpTy, SExtInst > m_SExt(const OpTy &Op)
Matches SExt.
is_zero m_Zero()
Match any null constant or a vector with all elements equal to 0.
auto m_ConstantInt()
Match an arbitrary ConstantInt and ignore it.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI KnownFPClass computeKnownFPClass(const Value *V, const APInt &DemandedElts, FPClassTest InterestedClasses, const SimplifyQuery &SQ, unsigned Depth=0)
Determine which floating-point classes are valid for V, and return them in KnownFPClass bit sets.
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
@ Known
Known to have no common set bits.
auto enumerate(FirstRange &&First, RestRanges &&...Rest)
Given two or more input ranges, returns a new range whose values are tuples (A, B,...
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
constexpr bool isMask_32(uint32_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
LLVM_ABI Constant * ConstantFoldCompareInstOperands(unsigned Predicate, Constant *LHS, Constant *RHS, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, const Instruction *I=nullptr)
Attempt to constant fold a compare instruction (icmp/fcmp) with the specified operands.
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
APFloat frexp(const APFloat &X, int &Exp, APFloat::roundingMode RM)
Equivalent of C standard library function.
auto dyn_cast_or_null(const Y &Val)
LLVM_READONLY APFloat maxnum(const APFloat &A, const APFloat &B)
Implements IEEE-754 2008 maxNum semantics.
constexpr unsigned MaxAnalysisRecursionDepth
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
APFloat scalbn(APFloat X, int Exp, APFloat::roundingMode RM)
Returns: X * 2^Exp for integral exponents.
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
constexpr int PoisonMaskElem
LLVM_ABI Value * findScalarElement(Value *V, unsigned EltNo)
Given a vector and an element number, see if the scalar value is already around as a register,...
@ NearestTiesToEven
roundTiesToEven.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
bool all_equal(std::initializer_list< T > Values)
Returns true if all Values in the initializer lists are equal or the list.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
LLVM_ABI Constant * ConstantFoldInstOperands(const Instruction *I, ArrayRef< Constant * > Ops, const DataLayout &DL, const TargetLibraryInfo *TLI=nullptr, bool AllowNonDeterministic=true)
ConstantFoldInstOperands - Attempt to constant fold an instruction with the specified operands.
constexpr uint64_t Make_64(uint32_t High, uint32_t Low)
Make a 64-bit integer from a high / low pair of 32-bit integers.
LLVM_ABI ConstantRange computeConstantRange(const Value *V, bool ForSigned, const SimplifyQuery &SQ, unsigned Depth=0)
Determine the possible constant range of an integer or vector of integer value.
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Represent subnormal handling kind for floating point instruction inputs and outputs.
bool isKnownNeverInfOrNaN() const
Return true if it's known this can never be an infinity or nan.
LLVM_ABI bool isKnownNeverLogicalZero(DenormalMode Mode) const
Return true if it's known this can never be interpreted as a zero.
SimplifyQuery getWithInstruction(const Instruction *I) const
LLVM_ABI bool isUndefValue(Value *V) const
If CanUseUndef is true, returns whether V is undef.