25#include "llvm/IR/IntrinsicsDirectX.h"
35#define DEBUG_TYPE "dxil-intrinsic-expansion"
50 if (IsRaw && M->getTargetTriple().getDXILVersion() >
VersionTuple(1, 2))
59 if (M->getTargetTriple().getDXILVersion() >=
VersionTuple(1, 9))
74 ConstantInt::get(IType, 0x7c00))
75 : ConstantInt::get(IType, 0x7c00);
82 ConstantInt::get(IType, 0xfc00))
83 : ConstantInt::get(IType, 0xfc00);
85 Value *IVal = Builder.CreateBitCast(Val, PosInf->
getType());
86 Value *B1 = Builder.CreateICmpEQ(IVal, PosInf);
87 Value *B2 = Builder.CreateICmpEQ(IVal, NegInf);
88 Value *B3 = Builder.CreateOr(B1, B2);
94 if (M->getTargetTriple().getDXILVersion() >=
VersionTuple(1, 9))
110 ConstantInt::get(IType, 0x7c00))
111 : ConstantInt::get(IType, 0x7c00);
117 ConstantInt::get(IType, 0x3ff))
118 : ConstantInt::get(IType, 0x3ff);
125 ConstantInt::get(IType, 0))
126 : ConstantInt::get(IType, 0);
128 Value *IVal = Builder.CreateBitCast(Val, ExpBitMask->
getType());
129 Value *Exp = Builder.CreateAnd(IVal, ExpBitMask);
130 Value *B1 = Builder.CreateICmpEQ(Exp, ExpBitMask);
132 Value *Sig = Builder.CreateAnd(IVal, SigBitMask);
133 Value *B2 = Builder.CreateICmpNE(Sig, Zero);
134 Value *B3 = Builder.CreateAnd(B1, B2);
140 if (M->getTargetTriple().getDXILVersion() >=
VersionTuple(1, 9))
156 ConstantInt::get(IType, 0x7c00))
157 : ConstantInt::get(IType, 0x7c00);
159 Value *IVal = Builder.CreateBitCast(Val, ExpBitMask->
getType());
160 Value *Exp = Builder.CreateAnd(IVal, ExpBitMask);
161 Value *B1 = Builder.CreateICmpNE(Exp, ExpBitMask);
167 if (M->getTargetTriple().getDXILVersion() >=
VersionTuple(1, 9))
183 ConstantInt::get(IType, 0x7c00))
184 : ConstantInt::get(IType, 0x7c00);
190 ConstantInt::get(IType, 0))
191 : ConstantInt::get(IType, 0);
193 Value *IVal = Builder.CreateBitCast(Val, ExpBitMask->
getType());
194 Value *Exp = Builder.CreateAnd(IVal, ExpBitMask);
195 Value *NotAllZeroes = Builder.CreateICmpNE(Exp, Zero);
196 Value *NotAllOnes = Builder.CreateICmpNE(Exp, ExpBitMask);
197 Value *B1 = Builder.CreateAnd(NotAllZeroes, NotAllOnes);
202 switch (
F.getIntrinsicID()) {
203 case Intrinsic::assume:
205 case Intrinsic::atan2:
206 case Intrinsic::fshl:
207 case Intrinsic::fshr:
209 case Intrinsic::is_fpclass:
211 case Intrinsic::log10:
213 case Intrinsic::powi:
214 case Intrinsic::dx_all:
215 case Intrinsic::dx_any:
216 case Intrinsic::dx_cross:
217 case Intrinsic::dx_uclamp:
218 case Intrinsic::dx_sclamp:
219 case Intrinsic::dx_nclamp:
220 case Intrinsic::dx_degrees:
221 case Intrinsic::dx_isinf:
222 case Intrinsic::dx_isnan:
223 case Intrinsic::dx_lerp:
224 case Intrinsic::dx_normalize:
225 case Intrinsic::dx_fdot:
226 case Intrinsic::dx_sdot:
227 case Intrinsic::dx_udot:
228 case Intrinsic::dx_sign:
229 case Intrinsic::dx_step:
230 case Intrinsic::dx_radians:
231 case Intrinsic::usub_sat:
232 case Intrinsic::vector_reduce_add:
233 case Intrinsic::vector_reduce_fadd:
234 case Intrinsic::matrix_multiply:
235 case Intrinsic::matrix_transpose:
236 case Intrinsic::umul_with_overflow:
237 case Intrinsic::smul_with_overflow:
239 case Intrinsic::dx_resource_load_rawbuffer:
241 F.getParent(),
F.getReturnType()->getStructElementType(0),
243 case Intrinsic::dx_resource_load_typedbuffer:
245 F.getParent(),
F.getReturnType()->getStructElementType(0),
247 case Intrinsic::dx_resource_store_rawbuffer:
249 F.getParent(),
F.getFunctionType()->getParamType(3),
true);
250 case Intrinsic::dx_resource_store_typedbuffer:
252 F.getParent(),
F.getFunctionType()->getParamType(2),
false);
260 Type *Ty =
A->getType();
264 Value *Cmp = Builder.CreateICmpULT(
A,
B,
"usub.cmp");
265 Value *
Sub = Builder.CreateSub(
A,
B,
"usub.sub");
266 Value *Zero = ConstantInt::get(Ty, 0);
267 return Builder.CreateSelect(Cmp, Zero,
Sub,
"usub.sat");
274 Type *Ty,
unsigned BW) {
275 assert(BW % 2 == 0 &&
"high-half split needs symmetric halves");
276 unsigned Half = BW / 2;
277 Value *HalfShift = ConstantInt::get(Ty, Half);
280 Value *U0 = Builder.CreateAnd(
A, LoMask);
281 Value *U1 = Builder.CreateLShr(
A, HalfShift);
282 Value *V0 = Builder.CreateAnd(
B, LoMask);
283 Value *
V1 = Builder.CreateLShr(
B, HalfShift);
285 Value *W0 = Builder.CreateMul(U0, V0);
286 Value *
T = Builder.CreateAdd(Builder.CreateMul(U1, V0),
287 Builder.CreateLShr(W0, HalfShift));
288 Value *W1 = Builder.CreateAnd(
T, LoMask);
289 Value *W2 = Builder.CreateLShr(
T, HalfShift);
290 W1 = Builder.CreateAdd(Builder.CreateMul(U0,
V1), W1);
291 return Builder.CreateAdd(Builder.CreateAdd(Builder.CreateMul(U1,
V1), W2),
292 Builder.CreateLShr(W1, HalfShift));
302 Type *Ty =
A->getType();
303 unsigned BW = Ty->getScalarSizeInBits();
313 Lo = Builder.CreateMul(
A,
B);
316 Signed ? Builder.CreateSExt(
A, WideTy) : Builder.CreateZExt(
A, WideTy);
318 Signed ? Builder.CreateSExt(
B, WideTy) : Builder.CreateZExt(
B, WideTy);
319 Value *Wide = Builder.CreateMul(WideA, WideB);
322 Ov = Builder.CreateICmpNE(Wide, Builder.CreateSExt(
Lo, WideTy));
324 Value *
Hi = Builder.CreateLShr(Wide, ConstantInt::get(WideTy, BW));
325 Ov = Builder.CreateICmpNE(
Hi, ConstantInt::get(WideTy, 0));
327 }
else if (BW == 32) {
331 Signed ? Intrinsic::dx_imul : Intrinsic::dx_umul;
332 Value *
Mul = Builder.CreateIntrinsic(ResTy, IntrinsicID, {
A,
B});
333 Value *
Hi = Builder.CreateExtractValue(
Mul, 0);
334 Lo = Builder.CreateExtractValue(
Mul, 1);
336 Ov = Builder.CreateICmpNE(
337 Hi, Builder.CreateAShr(
Lo, ConstantInt::get(Ty, BW - 1)));
339 Ov = Builder.CreateICmpNE(
Hi, ConstantInt::get(Ty, 0));
341 Lo = Builder.CreateMul(
A,
B);
346 Value *SignShift = ConstantInt::get(Ty, BW - 1);
347 Value *ASign = Builder.CreateAShr(
A, SignShift);
348 Value *BSign = Builder.CreateAShr(
B, SignShift);
349 Hi = Builder.CreateSub(
Hi, Builder.CreateAnd(ASign,
B));
350 Hi = Builder.CreateSub(
Hi, Builder.CreateAnd(BSign,
A));
351 Ov = Builder.CreateICmpNE(
Hi, Builder.CreateAShr(
Lo, SignShift));
353 Ov = Builder.CreateICmpNE(
Hi, ConstantInt::get(Ty, 0));
358 Agg = Builder.CreateInsertValue(Agg,
Lo, 0);
359 return Builder.CreateInsertValue(Agg, Ov, 1);
363 assert(IntrinsicId == Intrinsic::vector_reduce_add ||
364 IntrinsicId == Intrinsic::vector_reduce_fadd);
367 bool IsFAdd = (IntrinsicId == Intrinsic::vector_reduce_fadd);
370 Type *Ty =
X->getType();
372 unsigned XVecSize = XVec->getNumElements();
373 Value *Sum = Builder.CreateExtractElement(
X,
static_cast<uint64_t>(0));
379 Sum = Builder.CreateFAdd(Sum, StartValue);
383 for (
unsigned I = 1;
I < XVecSize;
I++) {
384 Value *Elt = Builder.CreateExtractElement(
X,
I);
386 Sum = Builder.CreateFAdd(Sum, Elt);
388 Sum = Builder.CreateAdd(Sum, Elt);
397 Type *Ty =
X->getType();
403 ConstantInt::get(EltTy, 0))
404 : ConstantInt::get(EltTy, 0);
405 auto *V = Builder.CreateSub(Zero,
X);
406 return Builder.CreateIntrinsic(Ty, Intrinsic::smax, {
X, V},
nullptr,
420 Value *op0_x = Builder.CreateExtractElement(op0, (
uint64_t)0,
"x0");
421 Value *op0_y = Builder.CreateExtractElement(op0, 1,
"x1");
422 Value *op0_z = Builder.CreateExtractElement(op0, 2,
"x2");
424 Value *op1_x = Builder.CreateExtractElement(op1, (
uint64_t)0,
"y0");
425 Value *op1_y = Builder.CreateExtractElement(op1, 1,
"y1");
426 Value *op1_z = Builder.CreateExtractElement(op1, 2,
"y2");
429 Value *xy = Builder.CreateFMul(x0, y1);
430 Value *yx = Builder.CreateFMul(y0, x1);
431 return Builder.CreateFSub(xy, yx, Orig->
getName());
434 Value *yz_zy = MulSub(op0_y, op0_z, op1_y, op1_z);
435 Value *zx_xz = MulSub(op0_z, op0_x, op1_z, op1_x);
436 Value *xy_yx = MulSub(op0_x, op0_y, op1_x, op1_y);
439 cross = Builder.CreateInsertElement(cross, yz_zy, (
uint64_t)0);
440 cross = Builder.CreateInsertElement(cross, zx_xz, 1);
441 cross = Builder.CreateInsertElement(cross, xy_yx, 2);
449 Type *ATy =
A->getType();
450 [[maybe_unused]]
Type *BTy =
B->getType();
460 int NumElts = AVec->getNumElements();
463 DotIntrinsic = Intrinsic::dx_dot2;
466 DotIntrinsic = Intrinsic::dx_dot3;
469 DotIntrinsic = Intrinsic::dx_dot4;
473 "Invalid dot product input vector: length is outside 2-4");
478 for (
int I = 0;
I < NumElts; ++
I)
479 Args.push_back(Builder.CreateExtractElement(
A, Builder.getInt32(
I)));
480 for (
int I = 0;
I < NumElts; ++
I)
481 Args.push_back(Builder.CreateExtractElement(
B, Builder.getInt32(
I)));
482 return Builder.CreateIntrinsic(ATy->
getScalarType(), DotIntrinsic, Args,
497 assert(DotIntrinsic == Intrinsic::dx_sdot ||
498 DotIntrinsic == Intrinsic::dx_udot);
501 Type *ATy =
A->getType();
502 [[maybe_unused]]
Type *BTy =
B->getType();
512 Intrinsic::ID MadIntrinsic = DotIntrinsic == Intrinsic::dx_sdot
514 : Intrinsic::dx_umad;
517 Result = Builder.CreateMul(Elt0, Elt1);
518 for (
unsigned I = 1;
I < AVec->getNumElements();
I++) {
519 Elt0 = Builder.CreateExtractElement(
A,
I);
520 Elt1 = Builder.CreateExtractElement(
B,
I);
521 Result = Builder.CreateIntrinsic(Result->getType(), MadIntrinsic,
531 Type *Ty =
X->getType();
539 Value *NewX = Builder.CreateFMul(Log2eConst,
X);
540 CallInst *Exp2Call = Builder.CreateIntrinsicWithoutFolding(
541 Ty, Intrinsic::exp2, {NewX},
nullptr,
"dx.exp2");
553 switch (TCI->getZExtValue()) {
567 Type *FTy =
F->getType();
568 unsigned FNumElem = 0;
574 Type *ElemTy = FVecTy->getElementType();
575 FNumElem = FVecTy->getNumElements();
576 BitWidth = ElemTy->getPrimitiveSizeInBits();
583 Value *FBitCast = Builder.CreateBitCast(
F, BitCastTy);
584 switch (TCI->getZExtValue()) {
591 Value *NegZeroSplat = Builder.CreateVectorSplat(FNumElem, NegZero);
593 Builder.CreateICmpEQ(FBitCast, NegZeroSplat,
"is.fpclass.negzero");
595 RetVal = Builder.CreateICmpEQ(FBitCast, NegZero,
"is.fpclass.negzero");
607 Type *Ty =
X->getType();
612 if (IntrinsicId == Intrinsic::dx_any)
613 return Builder.CreateOr(Result, Elt);
614 assert(IntrinsicId == Intrinsic::dx_all);
615 return Builder.CreateAnd(Result, Elt);
618 Value *Result =
nullptr;
619 if (!Ty->isVectorTy()) {
621 ? Builder.CreateFCmpUNE(
X, ConstantFP::get(EltTy, 0))
622 : Builder.CreateICmpNE(
X, ConstantInt::get(EltTy, 0));
627 ? Builder.CreateFCmpUNE(
630 ConstantFP::get(EltTy, 0)))
631 : Builder.CreateICmpNE(
634 ConstantInt::get(EltTy, 0)));
635 Result = Builder.CreateExtractElement(
Cond, (
uint64_t)0);
636 for (
unsigned I = 1;
I < XVec->getNumElements();
I++) {
637 Value *Elt = Builder.CreateExtractElement(
Cond,
I);
638 Result = ApplyOp(IntrinsicId, Result, Elt);
649 auto *V = Builder.CreateFSub(
Y,
X);
650 V = Builder.CreateFMul(S, V);
651 return Builder.CreateFAdd(
X, V,
"dx.lerp");
658 Type *Ty =
X->getType();
664 ConstantFP::get(EltTy, LogConstVal))
665 : ConstantFP::get(EltTy, LogConstVal);
666 CallInst *Log2Call = Builder.CreateIntrinsicWithoutFolding(
667 Ty, Intrinsic::log2, {
X},
nullptr,
"elt.log2");
670 return Builder.CreateFMul(Ln2Const, Log2Call);
687 const APFloat &fpVal = constantFP->getValueAPF();
691 return Builder.CreateFDiv(
X,
X);
699 const APFloat &fpVal = constantFP->getValueAPF();
704 Value *Multiplicand = Builder.CreateIntrinsic(EltTy, Intrinsic::dx_rsqrt,
706 nullptr,
"dx.rsqrt");
708 Value *MultiplicandVec =
709 Builder.CreateVectorSplat(XVec->getNumElements(), Multiplicand);
710 return Builder.CreateFMul(
X, MultiplicandVec);
716 Type *Ty =
X->getType();
720 Value *Tan = Builder.CreateFDiv(
Y,
X);
722 CallInst *Atan = Builder.CreateIntrinsicWithoutFolding(
723 Ty, Intrinsic::atan, {Tan},
nullptr,
"Elt.Atan");
731 Constant *Zero = ConstantFP::get(Ty, 0);
732 Value *AtanAddPi = Builder.CreateFAdd(Atan, Pi);
733 Value *AtanSubPi = Builder.CreateFSub(Atan, Pi);
736 Value *Result = Atan;
737 Value *XLt0 = Builder.CreateFCmpOLT(
X, Zero);
738 Value *XEq0 = Builder.CreateFCmpOEQ(
X, Zero);
739 Value *YGe0 = Builder.CreateFCmpOGE(
Y, Zero);
740 Value *YLt0 = Builder.CreateFCmpOLT(
Y, Zero);
743 Value *XLt0AndYGe0 = Builder.CreateAnd(XLt0, YGe0);
744 Result = Builder.CreateSelect(XLt0AndYGe0, AtanAddPi, Result);
747 Value *XLt0AndYLt0 = Builder.CreateAnd(XLt0, YLt0);
748 Result = Builder.CreateSelect(XLt0AndYLt0, AtanSubPi, Result);
751 Value *XEq0AndYLt0 = Builder.CreateAnd(XEq0, YLt0);
752 Result = Builder.CreateSelect(XEq0AndYLt0, NegHalfPi, Result);
755 Value *XEq0AndYGe0 = Builder.CreateAnd(XEq0, YGe0);
756 Result = Builder.CreateSelect(XEq0AndYGe0, HalfPi, Result);
761template <
bool LeftFunnel>
770 unsigned BitWidth = Ty->getScalarSizeInBits();
772 "Can't use Mask to compute modulo and inverse");
787 Constant *Mask = ConstantInt::get(Ty, Ty->getScalarSizeInBits() - 1);
792 Value *MaskedShift = Builder.CreateAnd(Shift, Mask);
797 Value *NotShift = Builder.CreateNot(Shift);
798 Value *InverseShift = Builder.CreateAnd(NotShift, Mask);
800 Constant *One = ConstantInt::get(Ty, 1);
805 ShiftedA = Builder.CreateShl(
A, MaskedShift);
806 Value *ShiftB1 = Builder.CreateLShr(
B, One);
807 ShiftedB = Builder.CreateLShr(ShiftB1, InverseShift);
809 Value *ShiftA1 = Builder.CreateShl(
A, One);
810 ShiftedA = Builder.CreateShl(ShiftA1, InverseShift);
811 ShiftedB = Builder.CreateLShr(
B, MaskedShift);
814 Value *Result = Builder.CreateOr(ShiftedA, ShiftedB);
822 Type *Ty =
X->getType();
825 if (IntrinsicId == Intrinsic::powi)
826 Y = Builder.CreateSIToFP(
Y, Ty);
829 Builder.CreateIntrinsic(Ty, Intrinsic::log2, {
X},
nullptr,
"elt.log2");
830 auto *
Mul = Builder.CreateFMul(Log2Call,
Y);
831 CallInst *Exp2Call = Builder.CreateIntrinsicWithoutFolding(
832 Ty, Intrinsic::exp2, {
Mul},
nullptr,
"elt.exp2");
842 Type *Ty =
X->getType();
845 Constant *One = ConstantFP::get(Ty->getScalarType(), 1.0);
846 Constant *Zero = ConstantFP::get(Ty->getScalarType(), 0.0);
849 if (Ty != Ty->getScalarType()) {
857 return Builder.CreateSelect(
Cond, Zero, One);
862 Type *Ty =
X->getType();
865 return Builder.CreateFMul(
X, PiOver180);
875 "Only expand double or int64 scalars or vectors");
876 bool IsVector =
false;
877 unsigned ExtractNum = 2;
879 ExtractNum = 2 * VT->getNumElements();
881 assert(IsRaw || ExtractNum == 4 &&
"TypedBufferLoad vector must be size 2");
890 while (ExtractNum > 0) {
891 unsigned LoadNum = std::min(ExtractNum, 4u);
895 Intrinsic::ID LoadIntrinsic = Intrinsic::dx_resource_load_typedbuffer;
898 LoadIntrinsic = Intrinsic::dx_resource_load_rawbuffer;
899 Value *Tmp = Builder.getInt32(4 *
Base * 2);
900 Args.push_back(Builder.CreateAdd(Orig->
getOperand(2), Tmp));
903 Value *
Load = Builder.CreateIntrinsic(LoadType, LoadIntrinsic, Args);
907 Value *Extract = Builder.CreateExtractValue(
Load, {0});
910 for (
unsigned I = 0;
I < LoadNum; ++
I)
912 Builder.CreateExtractElement(Extract, Builder.getInt32(
I)));
915 for (
unsigned I = 0;
I < LoadNum;
I += 2) {
916 Value *Combined =
nullptr;
919 Combined = Builder.CreateIntrinsic(
920 Builder.getDoubleTy(), Intrinsic::dx_asdouble,
921 {ExtractElements[I], ExtractElements[I + 1]});
926 Builder.CreateZExt(ExtractElements[
I], Builder.getInt64Ty());
928 Builder.CreateZExt(ExtractElements[
I + 1], Builder.getInt64Ty());
930 Value *ShiftedHi = Builder.CreateShl(
Hi, Builder.getInt64(32));
932 Combined = Builder.CreateOr(
Lo, ShiftedHi);
936 Result = Builder.CreateInsertElement(Result, Combined,
937 Builder.getInt32((
I / 2) +
Base));
942 ExtractNum -= LoadNum;
946 Value *CheckBit =
nullptr;
957 if (Indices[0] == 0) {
959 EVI->replaceAllUsesWith(Result);
962 assert(Indices[0] == 1 &&
"Unexpected type for typedbufferload");
967 for (
Value *L : Loads)
968 CheckBits.
push_back(Builder.CreateExtractValue(L, {1}));
969 CheckBit = Builder.CreateAnd(CheckBits);
971 EVI->replaceAllUsesWith(CheckBit);
973 EVI->eraseFromParent();
982 unsigned ValIndex = IsRaw ? 3 : 2;
987 "Only expand double or int64 scalars or vectors");
990 bool IsVector =
false;
991 unsigned ExtractNum = 2;
994 VecLen = VT->getNumElements();
995 assert(IsRaw || VecLen == 2 &&
"TypedBufferStore vector must be size 2");
996 ExtractNum = VecLen * 2;
1005 Type *SplitElementTy = Int32Ty;
1009 Value *LowBits =
nullptr;
1010 Value *HighBits =
nullptr;
1014 Value *Split = Builder.CreateIntrinsic(SplitTy, Intrinsic::dx_splitdouble,
1016 LowBits = Builder.CreateExtractValue(Split, 0);
1017 HighBits = Builder.CreateExtractValue(Split, 1);
1021 Constant *ShiftAmt = Builder.getInt64(32);
1027 LowBits = Builder.CreateTrunc(InputVal, SplitElementTy);
1028 Value *ShiftedVal = Builder.CreateLShr(InputVal, ShiftAmt);
1029 HighBits = Builder.CreateTrunc(ShiftedVal, SplitElementTy);
1034 for (
unsigned I = 0;
I < VecLen; ++
I) {
1036 Mask.push_back(
I + VecLen);
1038 Val = Builder.CreateShuffleVector(LowBits, HighBits, Mask);
1040 Val = Builder.CreateInsertElement(Val, LowBits, Builder.getInt32(0));
1041 Val = Builder.CreateInsertElement(Val, HighBits, Builder.getInt32(1));
1048 while (ExtractNum > 0) {
1049 unsigned StoreNum = std::min(ExtractNum, 4u);
1051 Intrinsic::ID StoreIntrinsic = Intrinsic::dx_resource_store_typedbuffer;
1054 StoreIntrinsic = Intrinsic::dx_resource_store_rawbuffer;
1055 Value *Tmp = Builder.getInt32(4 *
Base);
1056 Args.push_back(Builder.CreateAdd(Orig->
getOperand(2), Tmp));
1060 for (
unsigned I = 0;
I < StoreNum; ++
I) {
1061 Mask.push_back(
Base +
I);
1064 Value *SubVal = Val;
1066 SubVal = Builder.CreateShuffleVector(Val, Mask);
1068 Args.push_back(SubVal);
1070 Builder.CreateIntrinsic(Builder.getVoidTy(), StoreIntrinsic, Args);
1072 ExtractNum -= StoreNum;
1080 if (ClampIntrinsic == Intrinsic::dx_uclamp)
1081 return Intrinsic::umax;
1082 if (ClampIntrinsic == Intrinsic::dx_sclamp)
1083 return Intrinsic::smax;
1084 assert(ClampIntrinsic == Intrinsic::dx_nclamp);
1085 return Intrinsic::maxnum;
1089 if (ClampIntrinsic == Intrinsic::dx_uclamp)
1090 return Intrinsic::umin;
1091 if (ClampIntrinsic == Intrinsic::dx_sclamp)
1092 return Intrinsic::smin;
1093 assert(ClampIntrinsic == Intrinsic::dx_nclamp);
1094 return Intrinsic::minnum;
1102 Type *Ty =
X->getType();
1104 auto *MaxCall = Builder.CreateIntrinsic(Ty,
getMaxForClamp(ClampIntrinsic),
1105 {
X, Min},
nullptr,
"dx.max");
1106 return Builder.CreateIntrinsic(Ty,
getMinForClamp(ClampIntrinsic),
1107 {MaxCall, Max},
nullptr,
"dx.min");
1112 Type *Ty =
X->getType();
1115 return Builder.CreateFMul(
X, DegreesRatio);
1120 Type *Ty =
X->getType();
1130 GT = Builder.CreateFCmpOLT(Zero,
X);
1131 LT = Builder.CreateFCmpOLT(
X, Zero);
1134 GT = Builder.CreateICmpSLT(Zero,
X);
1135 LT = Builder.CreateICmpSLT(
X, Zero);
1138 Value *ZextGT = Builder.CreateZExt(GT, RetTy);
1139 Value *ZextLT = Builder.CreateZExt(LT, RetTy);
1141 return Builder.CreateSub(ZextGT, ZextLT);
1156 Type *EltTy = RetTy->getElementType();
1167 unsigned LHSSize = LHSRows * LHSCols;
1168 unsigned RHSSize = LHSCols * RHSCols;
1171 for (
unsigned I = 0;
I < LHSSize; ++
I)
1172 LHSElts[
I] = Builder.CreateExtractElement(
LHS,
I);
1173 for (
unsigned I = 0;
I < RHSSize; ++
I)
1174 RHSElts[
I] = Builder.CreateExtractElement(
RHS,
I);
1179 bool UseScalarFP = IsFP && (EltTy->
isDoubleTy() || LHSCols == 1);
1180 if (IsFP && !UseScalarFP) {
1183 FloatDotID = Intrinsic::dx_dot2;
1186 FloatDotID = Intrinsic::dx_dot3;
1189 FloatDotID = Intrinsic::dx_dot4;
1193 "Invalid matrix inner dimension for dot product: must be 2-4");
1198 for (
unsigned C = 0;
C < RHSCols; ++
C) {
1199 for (
unsigned R = 0; R < LHSRows; ++R) {
1202 for (
unsigned K = 0; K < LHSCols; ++K) {
1203 RowElts.
push_back(LHSElts[K * LHSRows + R]);
1210 Dot = Builder.CreateFMul(RowElts[0], ColElts[0]);
1211 for (
unsigned K = 1; K < LHSCols; ++K)
1212 Dot = Builder.CreateIntrinsic(EltTy, Intrinsic::fmuladd,
1213 {RowElts[K], ColElts[K], Dot});
1217 Args.append(RowElts.
begin(), RowElts.
end());
1218 Args.append(ColElts.
begin(), ColElts.
end());
1219 Dot = Builder.CreateIntrinsic(EltTy, FloatDotID, Args);
1222 Dot = Builder.CreateMul(RowElts[0], ColElts[0]);
1223 for (
unsigned K = 1; K < LHSCols; ++K)
1224 Dot = Builder.CreateIntrinsic(EltTy, Intrinsic::dx_imad,
1225 {RowElts[K], ColElts[K], Dot});
1227 unsigned ResIdx =
C * LHSRows + R;
1228 Result = Builder.CreateInsertElement(Result, Dot, ResIdx);
1242 unsigned NumElts = Rows * Cols;
1244 for (
unsigned I = 0;
I < NumElts; ++
I)
1245 Mask[
I] = (
I % Cols) * Rows + (
I / Cols);
1248 return Builder.CreateShuffleVector(Mat, Mask);
1252 Value *Result =
nullptr;
1254 switch (IntrinsicId) {
1255 case Intrinsic::abs:
1258 case Intrinsic::assume:
1261 case Intrinsic::atan2:
1264 case Intrinsic::fshl:
1267 case Intrinsic::fshr:
1270 case Intrinsic::exp:
1273 case Intrinsic::is_fpclass:
1276 case Intrinsic::log:
1279 case Intrinsic::log10:
1282 case Intrinsic::pow:
1283 case Intrinsic::powi:
1286 case Intrinsic::dx_all:
1287 case Intrinsic::dx_any:
1290 case Intrinsic::dx_cross:
1293 case Intrinsic::dx_uclamp:
1294 case Intrinsic::dx_sclamp:
1295 case Intrinsic::dx_nclamp:
1298 case Intrinsic::dx_degrees:
1301 case Intrinsic::dx_isinf:
1304 case Intrinsic::dx_isnan:
1307 case Intrinsic::dx_lerp:
1310 case Intrinsic::dx_normalize:
1313 case Intrinsic::dx_fdot:
1316 case Intrinsic::dx_sdot:
1317 case Intrinsic::dx_udot:
1320 case Intrinsic::dx_sign:
1323 case Intrinsic::dx_step:
1326 case Intrinsic::dx_radians:
1329 case Intrinsic::dx_resource_load_rawbuffer:
1333 case Intrinsic::dx_resource_store_rawbuffer:
1337 case Intrinsic::dx_resource_load_typedbuffer:
1341 case Intrinsic::dx_resource_store_typedbuffer:
1345 case Intrinsic::usub_sat:
1348 case Intrinsic::umul_with_overflow:
1349 case Intrinsic::smul_with_overflow:
1351 Intrinsic::smul_with_overflow);
1353 case Intrinsic::vector_reduce_add:
1354 case Intrinsic::vector_reduce_fadd:
1357 case Intrinsic::matrix_multiply:
1360 case Intrinsic::matrix_transpose:
1376 bool IntrinsicExpanded =
false;
1383 if (
F.user_empty() && IntrinsicExpanded)
1384 F.eraseFromParent();
1403 "DXIL Intrinsic Expansion",
false,
false)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
This file implements a class to represent arbitrary precision integral constant values and operations...
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static Value * expand16BitIsNormal(CallInst *Orig)
static Value * expandNormalizeIntrinsic(CallInst *Orig)
static Value * createMulHighUnsigned(IRBuilder<> &Builder, Value *A, Value *B, Type *Ty, unsigned BW)
static bool expandIntrinsic(Function &F, CallInst *Orig)
static Value * expandClampIntrinsic(CallInst *Orig, Intrinsic::ID ClampIntrinsic)
static Value * expand16BitIsInf(CallInst *Orig)
static bool expansionIntrinsics(Module &M)
static Value * expand16BitIsFinite(CallInst *Orig)
static Value * expandLerpIntrinsic(CallInst *Orig)
static Value * expandCrossIntrinsic(CallInst *Orig)
static Value * expandUsubSat(CallInst *Orig)
static Value * expandAnyOrAllIntrinsic(CallInst *Orig, Intrinsic::ID IntrinsicId)
static Value * expandMatrixTranspose(CallInst *Orig)
static Value * expandVecReduceAdd(CallInst *Orig, Intrinsic::ID IntrinsicId)
static Value * expandAtan2Intrinsic(CallInst *Orig)
static Value * expandLog10Intrinsic(CallInst *Orig)
static Intrinsic::ID getMinForClamp(Intrinsic::ID ClampIntrinsic)
static Value * expandStepIntrinsic(CallInst *Orig)
static Value * expandIntegerDotIntrinsic(CallInst *Orig, Intrinsic::ID DotIntrinsic)
static bool expandBufferStoreIntrinsic(CallInst *Orig, bool IsRaw)
static Value * expandLogIntrinsic(CallInst *Orig, float LogConstVal=numbers::ln2f)
static Value * expandDegreesIntrinsic(CallInst *Orig)
static Value * expandMulWithOverflow(CallInst *Orig, bool Signed)
static Value * expandPowIntrinsic(CallInst *Orig, Intrinsic::ID IntrinsicId)
static bool resourceAccessNeeds64BitExpansion(Module *M, Type *OverloadTy, bool IsRaw)
static Value * expandExpIntrinsic(CallInst *Orig)
static Value * expand16BitIsNaN(CallInst *Orig)
static Value * expandSignIntrinsic(CallInst *Orig)
static Intrinsic::ID getMaxForClamp(Intrinsic::ID ClampIntrinsic)
static Value * expandAbs(CallInst *Orig)
static Value * expandFloatDotIntrinsic(CallInst *Orig, Value *A, Value *B)
static Value * expandRadiansIntrinsic(CallInst *Orig)
static bool isIntrinsicExpansion(Function &F)
static bool expandBufferLoadIntrinsic(CallInst *Orig, bool IsRaw)
static Value * expandMatrixMultiply(CallInst *Orig)
static Value * expandIsFPClass(CallInst *Orig)
static Value * expandFunnelShiftIntrinsic(CallInst *Orig)
Module.h This file contains the declarations for the Module class.
This header defines various interfaces for pass management in LLVM.
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
const SmallVectorImpl< MachineOperand > & Cond
This file defines the SmallVector class.
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
bool runOnModule(Module &M) override
runOnModule - Virtual method overriden by subclasses to process the module being operated on.
DXILIntrinsicExpansionLegacy()
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
size_t size() const
Get the array size.
void setAttributes(AttributeList A)
Set the attributes for this call.
Value * getArgOperand(unsigned i) const
FunctionType * getFunctionType() const
AttributeList getAttributes() const
Return the attributes for this call.
This class represents a function call, abstracting a target machine's calling convention.
void setTailCall(bool IsTc=true)
static LLVM_ABI Constant * getSplat(ElementCount EC, Constant *Elt)
Return a ConstantVector with the specified constant in each element.
This is an important base class in LLVM.
bool isNullValue() const
Return true if this is the value that would be returned by getNullValue.
static LLVM_ABI Constant * getNullValue(Type *Ty)
Constructor to create a '0' constant of arbitrary type.
PreservedAnalyses run(Module &M, ModuleAnalysisManager &)
static constexpr ElementCount getFixed(ScalarTy MinVal)
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Type * getParamType(unsigned i) const
Parameter type accessors.
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI FastMathFlags getFastMathFlags() const LLVM_READONLY
Convenience function for getting all the fast-math flags, which must be an operator which supports th...
ModulePass class - This class is used to implement unstructured interprocedural optimizations and ana...
A Module instance is used to store all the information related to an LLVM module.
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
A set of analyses that are preserved following a run of a transformation pass.
static PreservedAnalyses none()
Convenience factory function for the empty preserved set.
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
static LLVM_ABI StructType * get(LLVMContext &Context, ArrayRef< Type * > Elements, bool isPacked=false)
This static method is the primary way to create a literal StructType.
The instances of the Type class are immutable: once they are created, they are never changed.
LLVM_ABI Type * getStructElementType(unsigned N) const
bool isVectorTy() const
True if this is an instance of VectorType.
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
LLVM_ABI Type * getWithNewBitWidth(unsigned NewBitWidth) const
Given an integer or vector type, change the lane bitwidth to NewBitwidth, whilst keeping the old numb...
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
bool isDoubleTy() const
Return true if this is 'double', a 64-bit IEEE fp type.
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
bool isIntegerTy() const
True if this is an instance of IntegerType.
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Value * getOperand(unsigned i) const
LLVM Value Representation.
Type * getType() const
All values are typed, get the type of this value.
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
iterator_range< user_iterator > users()
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
static LLVM_ABI VectorType * get(Type *ElementType, ElementCount EC)
This static method is the primary way to construct an VectorType.
Represents a version number in the form major[.minor[.subminor[.build]]].
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ C
The default llvm calling convention, compatible with C.
This is an optimization pass for GlobalISel generic memory operations.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
@ Load
The value being inserted comes from a load (InsertElement only).
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
ModulePass * createDXILIntrinsicExpansionLegacyPass()
Pass to expand intrinsic operations that lack DXIL opCodes.
@ Sub
Subtraction of integers.
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
AnalysisManager< Module > ModuleAnalysisManager
Convenience typedef for the Module analysis manager.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.