LLVM 24.0.0git
LoongArchISelLowering.cpp
Go to the documentation of this file.
1//=- LoongArchISelLowering.cpp - LoongArch DAG Lowering Implementation ---===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines the interfaces that LoongArch uses to lower LLVM code into
10// a selection DAG.
11//
12//===----------------------------------------------------------------------===//
13
15#include "LoongArch.h"
19#include "LoongArchSubtarget.h"
23#include "llvm/ADT/SmallSet.h"
24#include "llvm/ADT/Statistic.h"
30#include "llvm/IR/IRBuilder.h"
32#include "llvm/IR/IntrinsicsLoongArch.h"
34#include "llvm/Support/Debug.h"
39
40using namespace llvm;
41
42#define DEBUG_TYPE "loongarch-isel-lowering"
43
44STATISTIC(NumTailCalls, "Number of tail calls");
45
54
56 "loongarch-materialize-float-imm", cl::Hidden,
57 cl::desc("Maximum number of instructions used (including code sequence "
58 "to generate the value and moving the value to FPR) when "
59 "materializing floating-point immediates (default = 3)"),
61 cl::values(clEnumValN(NoMaterializeFPImm, "0", "Use constant pool"),
63 "Materialize FP immediate within 2 instructions"),
65 "Materialize FP immediate within 3 instructions"),
67 "Materialize FP immediate within 4 instructions"),
69 "Materialize FP immediate within 5 instructions"),
71 "Materialize FP immediate within 6 instructions "
72 "(behaves same as 5 on loongarch64)")));
73
74static cl::opt<bool> ZeroDivCheck("loongarch-check-zero-division", cl::Hidden,
75 cl::desc("Trap on integer division by zero."),
76 cl::init(false));
77
79 const LoongArchSubtarget &STI)
80 : TargetLowering(TM, STI), Subtarget(STI) {
81
82 MVT GRLenVT = Subtarget.getGRLenVT();
83
84 // Set up the register classes.
85
86 addRegisterClass(GRLenVT, &LoongArch::GPRRegClass);
87 if (Subtarget.hasBasicF())
88 addRegisterClass(MVT::f32, &LoongArch::FPR32RegClass);
89 if (Subtarget.hasBasicD())
90 addRegisterClass(MVT::f64, &LoongArch::FPR64RegClass);
91
92 static const MVT::SimpleValueType LSXVTs[] = {
93 MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64, MVT::v4f32, MVT::v2f64};
94 static const MVT::SimpleValueType LASXVTs[] = {
95 MVT::v32i8, MVT::v16i16, MVT::v8i32, MVT::v4i64, MVT::v8f32, MVT::v4f64};
96
97 if (Subtarget.hasExtLSX())
98 for (MVT VT : LSXVTs)
99 addRegisterClass(VT, &LoongArch::LSX128RegClass);
100
101 if (Subtarget.hasExtLASX())
102 for (MVT VT : LASXVTs)
103 addRegisterClass(VT, &LoongArch::LASX256RegClass);
104
105 // Set operations for LA32 and LA64.
106
108 MVT::i1, Promote);
109
116
119 GRLenVT, Custom);
120
122
127
129 setOperationAction(ISD::TRAP, MVT::Other, Legal);
130
134
136
137 // BITREV/REVB requires the 32S feature.
138 if (STI.has32S()) {
139 // Expand bitreverse.i16 with native-width bitrev and shift for now, before
140 // we get to know which of sll and revb.2h is faster.
143
144 // LA32 does not have REVB.2W and REVB.D due to the 64-bit operands, and
145 // the narrower REVB.W does not exist. But LA32 does have REVB.2H, so i16
146 // and i32 could still be byte-swapped relatively cheaply.
148 } else {
156 }
157
164
167
168 // Set operations for LA64 only.
169
170 if (Subtarget.is64Bit()) {
188
192 Custom);
194 }
195
196 // Set operations for LA32 only.
197
198 if (!Subtarget.is64Bit()) {
204 if (Subtarget.hasBasicD())
206 }
207
209
210 static const ISD::CondCode FPCCToExpand[] = {
213
214 // Set operations for 'F' feature.
215
216 if (Subtarget.hasBasicF()) {
217 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::f16, Expand);
218 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
219 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::bf16, Expand);
220 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
221 setCondCodeAction(FPCCToExpand, MVT::f32, Expand);
222
241 Subtarget.isSoftFPABI() ? LibCall : Custom);
243 Subtarget.isSoftFPABI() ? LibCall : Custom);
246 Subtarget.isSoftFPABI() ? LibCall : Custom);
249
250 if (Subtarget.is64Bit())
252
253 if (!Subtarget.hasBasicD()) {
255 if (Subtarget.is64Bit()) {
258 }
259 }
260 }
261
262 // Set operations for 'D' feature.
263
264 if (Subtarget.hasBasicD()) {
265 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f16, Expand);
266 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
267 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::bf16, Expand);
268 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
269 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
270 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
271 setCondCodeAction(FPCCToExpand, MVT::f64, Expand);
272
292 Subtarget.isSoftFPABI() ? LibCall : Custom);
295 Subtarget.isSoftFPABI() ? LibCall : Custom);
296
297 if (Subtarget.is64Bit())
299 }
300
301 // Set operations for 'LSX' feature.
302
303 if (Subtarget.hasExtLSX()) {
305 // Expand all truncating stores and extending loads.
306 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) {
307 setTruncStoreAction(VT, InnerVT, Expand);
310 setLoadExtAction(ISD::EXTLOAD, VT, InnerVT, Expand);
311 }
312 // By default everything must be expanded. Then we will selectively turn
313 // on ones that can be effectively codegen'd.
314 for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op)
316 }
317
318 for (MVT VT : LSXVTs) {
322
326
331 }
332 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32, MVT::v2i64}) {
335 Legal);
337 VT, Legal);
344 Expand);
359 }
360 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
362 for (MVT VT : {MVT::v8i16, MVT::v4i32, MVT::v2i64})
364 for (MVT VT : {MVT::v4i32, MVT::v2i64}) {
367 }
369 for (MVT VT : {MVT::v4f32, MVT::v2f64}) {
377 VT, Expand);
385 }
387 setOperationAction(ISD::FCEIL, {MVT::f32, MVT::f64}, Legal);
388 setOperationAction(ISD::FFLOOR, {MVT::f32, MVT::f64}, Legal);
389 setOperationAction(ISD::FTRUNC, {MVT::f32, MVT::f64}, Legal);
390 setOperationAction(ISD::FROUNDEVEN, {MVT::f32, MVT::f64}, Legal);
391
392 for (MVT VT :
393 {MVT::v16i8, MVT::v8i8, MVT::v4i8, MVT::v2i8, MVT::v8i16, MVT::v4i16,
394 MVT::v2i16, MVT::v4i32, MVT::v2i32, MVT::v2i64}) {
404 }
407 // We want to legalize this to an f64 load rather than an i64 load.
408 setOperationAction(ISD::LOAD, MVT::v2f32, Custom);
409 for (MVT VT : {MVT::v2i64, MVT::v4i32, MVT::v8i16})
411 for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64, MVT::v16i32, MVT::v8i64,
412 MVT::v16i64})
414 }
415
416 // Set operations for 'LASX' feature.
417
418 if (Subtarget.hasExtLASX()) {
419 for (MVT VT : LASXVTs) {
423
429
433 }
434 for (MVT VT : {MVT::v4i64, MVT::v8i32, MVT::v16i16, MVT::v32i8}) {
437 Legal);
439 VT, Legal);
446 Expand);
462 }
463 for (MVT VT : {MVT::v32i8, MVT::v16i16, MVT::v8i32})
465 for (MVT VT : {MVT::v16i16, MVT::v8i32, MVT::v4i64})
467 for (MVT VT : {MVT::v8i32, MVT::v4i32, MVT::v4i64}) {
471 }
472 for (MVT VT : {MVT::v8f32, MVT::v4f64}) {
480 VT, Expand);
488 }
491 for (MVT VT : {MVT::v4i64, MVT::v8i32, MVT::v16i16}) {
495 }
496 for (MVT VT :
497 {MVT::v2i64, MVT::v4i32, MVT::v4i64, MVT::v8i16, MVT::v8i32}) {
500 }
501 for (MVT VT : {MVT::v16i8, MVT::v8i16, MVT::v4i32})
503 }
504
505 // Set DAG combine for LA32 and LA64.
506 if (Subtarget.hasBasicF()) {
508 }
509
514
515 // On targets with the 32S feature, `select` is expanded into
516 // maskeqz + masknez + or (3 instructions), which is more expensive than
517 // on most other architectures where a single cmov-like instruction
518 // suffices. Enable a combine that can turn
519 // select cond, binop(X, Y), X -> binop X, (select cond, Y, 0)
520 // select cond, X, binop(X, Y) -> binop X, (select cond, 0, Y)
521 // for binop in {add, or, xor, sub}, replacing the 3-insn select (plus
522 // the original binop) with a single mask instruction plus the binop.
523 if (Subtarget.has32S())
525
526 // Set DAG combine for 'LSX' feature.
527
528 if (Subtarget.hasExtLSX()) {
540 }
541
542 // Set DAG combine for 'LASX' feature.
543 if (Subtarget.hasExtLASX()) {
546 }
547
548 // Compute derived properties from the register classes.
549 computeRegisterProperties(Subtarget.getRegisterInfo());
550
552
555
556 setMaxAtomicSizeInBitsSupported(Subtarget.getGRLen());
557
559
560 // Function alignments.
562 // Set preferred alignments.
563 setPrefFunctionAlignment(Subtarget.getPrefFunctionAlignment());
564 setPrefLoopAlignment(Subtarget.getPrefLoopAlignment());
565 setMaxBytesForAlignment(Subtarget.getMaxBytesForAlignment());
566
567 // cmpxchg sizes down to 8 bits become legal if LAMCAS is available.
568 if (Subtarget.hasLAMCAS())
570
571 if (Subtarget.hasSCQ()) {
574 }
575
576 // Disable strict node mutation.
577 IsStrictFPEnabled = true;
578}
579
581 const GlobalAddressSDNode *GA) const {
582 // In order to maximise the opportunity for common subexpression elimination,
583 // keep a separate ADD node for the global address offset instead of folding
584 // it in the global address node. Later peephole optimisations may choose to
585 // fold it back in when profitable.
586 return false;
587}
588
590 SelectionDAG &DAG) const {
591 switch (Op.getOpcode()) {
593 return lowerATOMIC_FENCE(Op, DAG);
595 return lowerEH_DWARF_CFA(Op, DAG);
597 return lowerGlobalAddress(Op, DAG);
599 return lowerGlobalTLSAddress(Op, DAG);
601 return lowerINTRINSIC_WO_CHAIN(Op, DAG);
603 return lowerINTRINSIC_W_CHAIN(Op, DAG);
605 return lowerINTRINSIC_VOID(Op, DAG);
607 return lowerBlockAddress(Op, DAG);
608 case ISD::JumpTable:
609 return lowerJumpTable(Op, DAG);
610 case ISD::SHL_PARTS:
611 return lowerShiftLeftParts(Op, DAG);
612 case ISD::SRA_PARTS:
613 return lowerShiftRightParts(Op, DAG, true);
614 case ISD::SRL_PARTS:
615 return lowerShiftRightParts(Op, DAG, false);
617 return lowerConstantPool(Op, DAG);
618 case ISD::FP_TO_SINT:
619 return lowerFP_TO_SINT(Op, DAG);
620 case ISD::FP_TO_UINT:
621 return lowerFP_TO_UINT(Op, DAG);
622 case ISD::BITCAST:
623 return lowerBITCAST(Op, DAG);
624 case ISD::UINT_TO_FP:
625 return lowerUINT_TO_FP(Op, DAG);
626 case ISD::SINT_TO_FP:
627 return lowerSINT_TO_FP(Op, DAG);
628 case ISD::VASTART:
629 return lowerVASTART(Op, DAG);
630 case ISD::FRAMEADDR:
631 return lowerFRAMEADDR(Op, DAG);
632 case ISD::RETURNADDR:
633 return lowerRETURNADDR(Op, DAG);
635 return lowerSET_ROUNDING(Op, DAG);
637 return lowerGET_ROUNDING(Op, DAG);
639 return lowerWRITE_REGISTER(Op, DAG);
641 return lowerINSERT_VECTOR_ELT(Op, DAG);
643 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
645 return lowerBUILD_VECTOR(Op, DAG);
647 return lowerCONCAT_VECTORS(Op, DAG);
649 return lowerVECTOR_SHUFFLE(Op, DAG);
650 case ISD::BITREVERSE:
651 return lowerBITREVERSE(Op, DAG);
653 return lowerSCALAR_TO_VECTOR(Op, DAG);
654 case ISD::PREFETCH:
655 return lowerPREFETCH(Op, DAG);
656 case ISD::SELECT:
657 return lowerSELECT(Op, DAG);
658 case ISD::BRCOND:
659 return lowerBRCOND(Op, DAG);
660 case ISD::FP_TO_FP16:
661 return lowerFP_TO_FP16(Op, DAG);
662 case ISD::FP16_TO_FP:
663 return lowerFP16_TO_FP(Op, DAG);
664 case ISD::FP_TO_BF16:
665 return lowerFP_TO_BF16(Op, DAG);
666 case ISD::BF16_TO_FP:
667 return lowerBF16_TO_FP(Op, DAG);
669 return lowerVECREDUCE_ADD(Op, DAG);
670 case ISD::ROTL:
671 case ISD::ROTR:
672 return lowerRotate(Op, DAG);
680 return lowerVECREDUCE(Op, DAG);
681 case ISD::ConstantFP:
682 return lowerConstantFP(Op, DAG);
683 case ISD::SETCC:
684 return lowerSETCC(Op, DAG);
685 case ISD::FP_ROUND:
686 return lowerFP_ROUND(Op, DAG);
687 case ISD::FP_EXTEND:
688 return lowerFP_EXTEND(Op, DAG);
690 return lowerSIGN_EXTEND_VECTOR_INREG(Op, DAG);
692 return lowerDYNAMIC_STACKALLOC(Op, DAG);
693 case ISD::ANY_EXTEND:
694 return lowerANY_EXTEND(Op, DAG);
695 }
696 return SDValue();
697}
698
699// Helper to attempt to return a cheaper, bit-inverted version of \p V.
701 // TODO: don't always ignore oneuse constraints.
702 V = peekThroughBitcasts(V);
703 EVT VT = V.getValueType();
704
705 // Match not(xor X, -1) -> X.
706 if (V.getOpcode() == ISD::XOR &&
707 (ISD::isBuildVectorAllOnes(V.getOperand(1).getNode()) ||
708 isAllOnesConstant(V.getOperand(1))))
709 return V.getOperand(0);
710
711 // Match not(extract_subvector(not(X)) -> extract_subvector(X).
712 if (V.getOpcode() == ISD::EXTRACT_SUBVECTOR &&
713 (isNullConstant(V.getOperand(1)) || V.getOperand(0).hasOneUse())) {
714 if (SDValue Not = isNOT(V.getOperand(0), DAG)) {
715 Not = DAG.getBitcast(V.getOperand(0).getValueType(), Not);
716 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, SDLoc(Not), VT, Not,
717 V.getOperand(1));
718 }
719 }
720
721 // Match not(SplatVector(not(X)) -> SplatVector(X).
722 if (V.getOpcode() == ISD::BUILD_VECTOR) {
723 if (SDValue SplatValue =
724 cast<BuildVectorSDNode>(V.getNode())->getSplatValue()) {
725 if (!V->isOnlyUserOf(SplatValue.getNode()))
726 return SDValue();
727
728 if (SDValue Not = isNOT(SplatValue, DAG)) {
729 Not = DAG.getBitcast(V.getOperand(0).getValueType(), Not);
730 return DAG.getSplat(VT, SDLoc(Not), Not);
731 }
732 }
733 }
734
735 // Match not(or(not(X),not(Y))) -> and(X, Y).
736 if (V.getOpcode() == ISD::OR && DAG.getTargetLoweringInfo().isTypeLegal(VT) &&
737 V.getOperand(0).hasOneUse() && V.getOperand(1).hasOneUse()) {
738 // TODO: Handle cases with single NOT operand -> VANDN
739 if (SDValue Op1 = isNOT(V.getOperand(1), DAG))
740 if (SDValue Op0 = isNOT(V.getOperand(0), DAG))
741 return DAG.getNode(ISD::AND, SDLoc(V), VT, DAG.getBitcast(VT, Op0),
742 DAG.getBitcast(VT, Op1));
743 }
744
745 // TODO: Add more matching patterns. Such as,
746 // not(concat_vectors(not(X), not(Y))) -> concat_vectors(X, Y).
747 // not(slt(C, X)) -> slt(X - 1, C)
748 return SDValue();
749}
750
751// Combine two ISD::FP_ROUND / LoongArchISD::VFCVT nodes with same type to
752// LoongArchISD::VFCVT. For example:
753// x1 = fp_round x, 0
754// y1 = fp_round y, 0
755// z = concat_vectors x1, y1
756// Or
757// x1 = LoongArch::VFCVT undef, x
758// y1 = LoongArch::VFCVT undef, y
759// z = LoongArchISD::VPACKEV y1, x1; or LoongArchISD::VPERMI y1, x1, 68
760// can be combined to:
761// z = LoongArch::VFCVT y, x
763 const LoongArchSubtarget &Subtarget) {
764 assert(((N->getOpcode() == ISD::CONCAT_VECTORS && N->getNumOperands() == 2) ||
765 (N->getOpcode() == LoongArchISD::VPACKEV) ||
766 (N->getOpcode() == LoongArchISD::VPERMI)) &&
767 "Invalid Node");
768
769 SDValue Op0 = peekThroughBitcasts(N->getOperand(0));
770 SDValue Op1 = peekThroughBitcasts(N->getOperand(1));
771 unsigned Opcode0 = Op0.getOpcode();
772 unsigned Opcode1 = Op1.getOpcode();
773 if (Opcode0 != Opcode1)
774 return SDValue();
775
776 if (Opcode0 != ISD::FP_ROUND && Opcode0 != LoongArchISD::VFCVT)
777 return SDValue();
778
779 // Check if two nodes have only one use.
780 if (!Op0.hasOneUse() || !Op1.hasOneUse())
781 return SDValue();
782
783 EVT VT = N.getValueType();
784 EVT SVT0 = Op0.getValueType();
785 EVT SVT1 = Op1.getValueType();
786 // Check if two nodes have the same result type.
787 if (SVT0 != SVT1)
788 return SDValue();
789
790 // Check if two nodes have the same operand type.
791 EVT SSVT0 = Op0.getOperand(0).getValueType();
792 EVT SSVT1 = Op1.getOperand(0).getValueType();
793 if (SSVT0 != SSVT1)
794 return SDValue();
795
796 if (N->getOpcode() == ISD::CONCAT_VECTORS && Opcode0 == ISD::FP_ROUND) {
797 if (Subtarget.hasExtLASX() && VT.is256BitVector() && SVT0 == MVT::v4f32 &&
798 SSVT0 == MVT::v4f64) {
799 // A vector_shuffle is required in the final step, as xvfcvt instruction
800 // operates on each 128-bit segament as a lane.
801 SDValue Res = DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v8f32,
802 Op1.getOperand(0), Op0.getOperand(0));
803 SDValue Undef = DAG.getUNDEF(Res.getValueType());
804 // After VFCVT, the high part of Res comes from the high parts of Op0 and
805 // Op1, and the low part comes from the low parts of Op0 and Op1. However,
806 // the desired order requires Op0 to fully occupy the lower half and Op1
807 // the upper half of Res. The Mask reorders the elements of Res to achieve
808 // this:
809 // - The first four elements (0, 1, 4, 5) come from Op0.
810 // - The next four elements (2, 3, 6, 7) come from Op1.
811 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
812 Res = DAG.getVectorShuffle(Res.getValueType(), DL, Res, Undef, Mask);
813 return DAG.getBitcast(VT, Res);
814 }
815 }
816
817 if ((N->getOpcode() == LoongArchISD::VPACKEV ||
818 N->getOpcode() == LoongArchISD::VPERMI) &&
819 Opcode0 == LoongArchISD::VFCVT) {
820 // For VPACKEV or VPERMI, check if the first operation of VFCVT is undef.
821 if (!Op0.getOperand(0).isUndef() || !Op1.getOperand(0).isUndef())
822 return SDValue();
823
824 if (!Subtarget.hasExtLSX() || SVT0 != MVT::v4f32 || SSVT0 != MVT::v2f64)
825 return SDValue();
826
827 if (N->getOpcode() == LoongArchISD::VPACKEV &&
828 (VT == MVT::v2i64 || VT == MVT::v2f64)) {
829 SDValue Res = DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32,
830 Op0.getOperand(1), Op1.getOperand(1));
831 return DAG.getBitcast(VT, Res);
832 }
833
834 if (N->getOpcode() == LoongArchISD::VPERMI && VT == MVT::v4f32) {
835 int64_t Imm = cast<ConstantSDNode>(N->getOperand(2))->getSExtValue();
836 if (Imm != 68)
837 return SDValue();
838 return DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32, Op0.getOperand(1),
839 Op1.getOperand(1));
840 }
841 }
842
843 return SDValue();
844}
845
846SDValue LoongArchTargetLowering::lowerFP_ROUND(SDValue Op,
847 SelectionDAG &DAG) const {
848 SDLoc DL(Op);
849 SDValue In = Op.getOperand(0);
850 MVT VT = Op.getSimpleValueType();
851 MVT SVT = In.getSimpleValueType();
852
853 if (VT == MVT::v4f32 && SVT == MVT::v4f64) {
854 SDValue Lo, Hi;
855 std::tie(Lo, Hi) = DAG.SplitVector(In, DL);
856 return DAG.getNode(LoongArchISD::VFCVT, DL, VT, Hi, Lo);
857 }
858
859 return SDValue();
860}
861
862SDValue LoongArchTargetLowering::lowerFP_EXTEND(SDValue Op,
863 SelectionDAG &DAG) const {
864
865 SDLoc DL(Op);
866 EVT VT = Op.getValueType();
867 SDValue Src = Op->getOperand(0);
868 EVT SVT = Src.getValueType();
869
870 bool V2F32ToV2F64 =
871 VT == MVT::v2f64 && SVT == MVT::v2f32 && Subtarget.hasExtLSX();
872 bool V4F32ToV4F64 =
873 VT == MVT::v4f64 && SVT == MVT::v4f32 && Subtarget.hasExtLASX();
874 if (!V2F32ToV2F64 && !V4F32ToV4F64)
875 return SDValue();
876
877 // Check if Op is the high part of vector.
878 auto CheckVecHighPart = [](SDValue Op) {
880 if (Op.getOpcode() == ISD::EXTRACT_SUBVECTOR) {
881 SDValue SOp = Op.getOperand(0);
882 EVT SVT = SOp.getValueType();
883 if (!SVT.isVector() || (SVT.getVectorNumElements() % 2 != 0))
884 return SDValue();
885
886 const uint64_t Imm = Op.getConstantOperandVal(1);
887 if (Imm == SVT.getVectorNumElements() / 2)
888 return SOp;
889 return SDValue();
890 }
891 return SDValue();
892 };
893
894 unsigned Opcode;
895 SDValue VFCVTOp;
896 EVT WideOpVT = SVT.getSimpleVT().getDoubleNumVectorElementsVT();
897 SDValue ZeroIdx = DAG.getVectorIdxConstant(0, DL);
898
899 // If the operand of ISD::FP_EXTEND comes from the high part of vector,
900 // generate LoongArchISD::VFCVTH, otherwise LoongArchISD::VFCVTL.
901 if (SDValue V = CheckVecHighPart(Src)) {
902 assert(V.getValueSizeInBits() == WideOpVT.getSizeInBits() &&
903 "Unexpected wide vector");
904 Opcode = LoongArchISD::VFCVTH;
905 VFCVTOp = DAG.getBitcast(WideOpVT, V);
906 } else {
907 Opcode = LoongArchISD::VFCVTL;
908 VFCVTOp = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, WideOpVT,
909 DAG.getUNDEF(WideOpVT), Src, ZeroIdx);
910 }
911
912 // v2f64 = fp_extend v2f32
913 if (V2F32ToV2F64)
914 return DAG.getNode(Opcode, DL, VT, VFCVTOp);
915
916 // v4f64 = fp_extend v4f32
917 if (V4F32ToV4F64) {
918 // XVFCVT instruction operates on each 128-bit segment as a lane, so a
919 // vector_shuffle is required firstly.
920 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
921 SDValue Res = DAG.getVectorShuffle(WideOpVT, DL, VFCVTOp,
922 DAG.getUNDEF(WideOpVT), Mask);
923 Res = DAG.getNode(Opcode, DL, VT, Res);
924 return Res;
925 }
926
927 return SDValue();
928}
929
930SDValue LoongArchTargetLowering::lowerConstantFP(SDValue Op,
931 SelectionDAG &DAG) const {
932 EVT VT = Op.getValueType();
933 ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Op);
934 const APFloat &FPVal = CFP->getValueAPF();
935 SDLoc DL(CFP);
936
937 assert((VT == MVT::f32 && Subtarget.hasBasicF()) ||
938 (VT == MVT::f64 && Subtarget.hasBasicD()));
939
940 // If value is 0.0 or -0.0, just ignore it.
941 if (FPVal.isZero())
942 return SDValue();
943
944 // If lsx enabled, use cheaper 'vldi' instruction if possible.
945 if (isFPImmVLDILegal(FPVal, VT))
946 return SDValue();
947
948 // Construct as integer, and move to float register.
949 APInt INTVal = FPVal.bitcastToAPInt();
950
951 // If more than MaterializeFPImmInsNum instructions will be used to
952 // generate the INTVal and move it to float register, fallback to
953 // use floating point load from the constant pool.
955 int InsNum = Seq.size() + ((VT == MVT::f64 && !Subtarget.is64Bit()) ? 2 : 1);
956 if (InsNum > MaterializeFPImmInsNum && !FPVal.isOne())
957 return SDValue();
958
959 switch (VT.getSimpleVT().SimpleTy) {
960 default:
961 llvm_unreachable("Unexpected floating point type!");
962 break;
963 case MVT::f32: {
964 SDValue NewVal = DAG.getConstant(INTVal, DL, MVT::i32);
965 if (Subtarget.is64Bit())
966 NewVal = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, NewVal);
967 return DAG.getNode(Subtarget.is64Bit() ? LoongArchISD::MOVGR2FR_W_LA64
968 : LoongArchISD::MOVGR2FR_W,
969 DL, VT, NewVal);
970 }
971 case MVT::f64: {
972 if (Subtarget.is64Bit()) {
973 SDValue NewVal = DAG.getConstant(INTVal, DL, MVT::i64);
974 return DAG.getNode(LoongArchISD::MOVGR2FR_D, DL, VT, NewVal);
975 }
976 SDValue Lo = DAG.getConstant(INTVal.trunc(32), DL, MVT::i32);
977 SDValue Hi = DAG.getConstant(INTVal.lshr(32).trunc(32), DL, MVT::i32);
978 return DAG.getNode(LoongArchISD::MOVGR2FR_D_LO_HI, DL, VT, Lo, Hi);
979 }
980 }
981
982 return SDValue();
983}
984
985// Ensure SETCC result and operand have the same bit width; isel does not
986// support mismatched widths.
987SDValue LoongArchTargetLowering::lowerSETCC(SDValue Op,
988 SelectionDAG &DAG) const {
989 SDLoc DL(Op);
990 EVT ResultVT = Op.getValueType();
991 EVT OperandVT = Op.getOperand(0).getValueType();
992
993 EVT SetCCResultVT =
994 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), OperandVT);
995
996 if (ResultVT == SetCCResultVT)
997 return Op;
998
999 assert(Op.getOperand(0).getValueType() == Op.getOperand(1).getValueType() &&
1000 "SETCC operands must have the same type!");
1001
1002 SDValue SetCCNode =
1003 DAG.getNode(ISD::SETCC, DL, SetCCResultVT, Op.getOperand(0),
1004 Op.getOperand(1), Op.getOperand(2));
1005
1006 if (ResultVT.bitsGT(SetCCResultVT))
1007 SetCCNode = DAG.getNode(ISD::SIGN_EXTEND, DL, ResultVT, SetCCNode);
1008 else if (ResultVT.bitsLT(SetCCResultVT))
1009 SetCCNode = DAG.getNode(ISD::TRUNCATE, DL, ResultVT, SetCCNode);
1010
1011 return SetCCNode;
1012}
1013
1014// Lower sext_invec using vslti instructions.
1015// For example:
1016// %b = sext <4 x i16> %a to <4 x i32>
1017// can be lowered to:
1018// VSLTI_H vr2, vr1, 0
1019// VILVL.H vr1, vr2, vr1
1020SDValue LoongArchTargetLowering::lowerSIGN_EXTEND_VECTOR_INREG(
1021 SDValue Op, SelectionDAG &DAG) const {
1022 SDLoc DL(Op);
1023 SDValue Src = Op.getOperand(0);
1024 MVT SrcVT = Src.getSimpleValueType();
1025 MVT DstVT = Op.getSimpleValueType();
1026
1027 if (!SrcVT.is128BitVector())
1028 return SDValue();
1029
1030 // lower to VSLTI + VILVL if extend could be done in single step.
1031 if (DstVT.getScalarSizeInBits() / SrcVT.getScalarSizeInBits() == 2) {
1032 SDValue Zero = DAG.getConstant(0, DL, SrcVT);
1033 SDValue Mask = DAG.getNode(ISD::SETCC, DL, SrcVT, Src, Zero,
1034 DAG.getCondCode(ISD::SETLT));
1035 SDValue LoInterleaved =
1036 DAG.getNode(LoongArchISD::VILVL, DL, SrcVT, Mask, Src);
1037
1038 return DAG.getBitcast(DstVT, LoInterleaved);
1039 }
1040
1041 return SDValue();
1042}
1043
1044// ANY_EXTEND can be replaced by ZERO_EXTEND when LASX is enabled.
1045SDValue LoongArchTargetLowering::lowerANY_EXTEND(SDValue Op,
1046 SelectionDAG &DAG) const {
1047 assert(Subtarget.hasExtLASX());
1048 // We don't have corresponding instrunction for ANY_EXTEND, lowering it to
1049 // ZERO_EXTEND won't break its semantics, while avoid scalar extract/insert.
1050 return DAG.getNode(ISD::ZERO_EXTEND, SDLoc(Op), Op.getValueType(),
1051 Op.getOperand(0));
1052}
1053
1054// Lower vecreduce_add using vhaddw instructions.
1055// For Example:
1056// call i32 @llvm.vector.reduce.add.v4i32(<4 x i32> %a)
1057// can be lowered to:
1058// VHADDW_D_W vr0, vr0, vr0
1059// VHADDW_Q_D vr0, vr0, vr0
1060// VPICKVE2GR_D a0, vr0, 0
1061// ADDI_W a0, a0, 0
1062SDValue LoongArchTargetLowering::lowerVECREDUCE_ADD(SDValue Op,
1063 SelectionDAG &DAG) const {
1064
1065 SDLoc DL(Op);
1066 MVT OpVT = Op.getSimpleValueType();
1067 SDValue Val = Op.getOperand(0);
1068
1069 unsigned NumEles = Val.getSimpleValueType().getVectorNumElements();
1070 unsigned EleBits = Val.getSimpleValueType().getScalarSizeInBits();
1071 unsigned ResBits = OpVT.getScalarSizeInBits();
1072
1073 unsigned LegalVecSize = 128;
1074 bool isLASX256Vector =
1075 Subtarget.hasExtLASX() && Val.getValueSizeInBits() == 256;
1076
1077 // Ensure operand type legal or enable it legal.
1078 while (!isTypeLegal(Val.getSimpleValueType())) {
1079 Val = DAG.WidenVector(Val, DL);
1080 }
1081
1082 // NumEles is designed for iterations count, v4i32 for LSX
1083 // and v8i32 for LASX should have the same count.
1084 if (isLASX256Vector) {
1085 NumEles /= 2;
1086 LegalVecSize = 256;
1087 }
1088
1089 EleBits *= 2;
1090 for (unsigned i = 1; i < NumEles; i *= 2, EleBits *= 2) {
1091 EleBits = std::min(EleBits, 64u);
1092 MVT IntTy = MVT::getIntegerVT(EleBits);
1093 MVT VecTy = MVT::getVectorVT(IntTy, LegalVecSize / EleBits);
1094 Val = DAG.getNode(LoongArchISD::VHADDW, DL, VecTy, Val, Val);
1095 }
1096
1097 if (isLASX256Vector) {
1098 SDValue Tmp = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, Val,
1099 DAG.getConstant(2, DL, Subtarget.getGRLenVT()));
1100 Val = DAG.getNode(ISD::ADD, DL, MVT::v4i64, Tmp, Val);
1101 }
1102
1103 Val = DAG.getBitcast(MVT::getVectorVT(OpVT, LegalVecSize / ResBits), Val);
1104 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, OpVT, Val,
1105 DAG.getConstant(0, DL, Subtarget.getGRLenVT()));
1106}
1107
1108// Lower vecreduce_and/or/xor/[s/u]max/[s/u]min.
1109// For Example:
1110// call i32 @llvm.vector.reduce.smax.v4i32(<4 x i32> %a)
1111// can be lowered to:
1112// VBSRL_V vr1, vr0, 8
1113// VMAX_W vr0, vr1, vr0
1114// VBSRL_V vr1, vr0, 4
1115// VMAX_W vr0, vr1, vr0
1116// VPICKVE2GR_W a0, vr0, 0
1117// For 256 bit vector, it is illegal and will be spilt into
1118// two 128 bit vector by default then processed by this.
1119SDValue LoongArchTargetLowering::lowerVECREDUCE(SDValue Op,
1120 SelectionDAG &DAG) const {
1121 SDLoc DL(Op);
1122
1123 MVT OpVT = Op.getSimpleValueType();
1124 SDValue Val = Op.getOperand(0);
1125
1126 unsigned NumEles = Val.getSimpleValueType().getVectorNumElements();
1127 unsigned EleBits = Val.getSimpleValueType().getScalarSizeInBits();
1128
1129 // Ensure operand type legal or enable it legal.
1130 while (!isTypeLegal(Val.getSimpleValueType())) {
1131 Val = DAG.WidenVector(Val, DL);
1132 }
1133
1134 unsigned Opcode = ISD::getVecReduceBaseOpcode(Op.getOpcode());
1135 MVT VecTy = Val.getSimpleValueType();
1136 MVT GRLenVT = Subtarget.getGRLenVT();
1137
1138 for (int i = NumEles; i > 1; i /= 2) {
1139 SDValue ShiftAmt = DAG.getConstant(i * EleBits / 16, DL, GRLenVT);
1140 SDValue Tmp = DAG.getNode(LoongArchISD::VBSRL, DL, VecTy, Val, ShiftAmt);
1141 Val = DAG.getNode(Opcode, DL, VecTy, Tmp, Val);
1142 }
1143
1144 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, OpVT, Val,
1145 DAG.getConstant(0, DL, GRLenVT));
1146}
1147
1148SDValue LoongArchTargetLowering::lowerPREFETCH(SDValue Op,
1149 SelectionDAG &DAG) const {
1150 unsigned IsData = Op.getConstantOperandVal(4);
1151
1152 // We don't support non-data prefetch.
1153 // Just preserve the chain.
1154 if (!IsData)
1155 return Op.getOperand(0);
1156
1157 return Op;
1158}
1159
1160SDValue LoongArchTargetLowering::lowerRotate(SDValue Op,
1161 SelectionDAG &DAG) const {
1162 MVT VT = Op.getSimpleValueType();
1163 assert(VT.isVector() && "Unexpected type");
1164
1165 SDLoc DL(Op);
1166 SDValue R = Op.getOperand(0);
1167 SDValue Amt = Op.getOperand(1);
1168 unsigned Opcode = Op.getOpcode();
1169 unsigned EltSizeInBits = VT.getScalarSizeInBits();
1170
1171 auto checkCstSplat = [](SDValue V, APInt &CstSplatValue) {
1172 if (V.getOpcode() != ISD::BUILD_VECTOR)
1173 return false;
1174 if (SDValue SplatValue =
1175 cast<BuildVectorSDNode>(V.getNode())->getSplatValue()) {
1176 if (auto *C = dyn_cast<ConstantSDNode>(SplatValue)) {
1177 CstSplatValue = C->getAPIntValue();
1178 return true;
1179 }
1180 }
1181 return false;
1182 };
1183
1184 // Check for constant splat rotation amount.
1185 APInt CstSplatValue;
1186 bool IsCstSplat = checkCstSplat(Amt, CstSplatValue);
1187 bool isROTL = Opcode == ISD::ROTL;
1188
1189 // Check for splat rotate by zero.
1190 if (IsCstSplat && CstSplatValue.urem(EltSizeInBits) == 0)
1191 return R;
1192
1193 // LoongArch targets always prefer ISD::ROTR.
1194 if (isROTL) {
1195 SDValue Zero = DAG.getConstant(0, DL, VT);
1196 return DAG.getNode(ISD::ROTR, DL, VT, R,
1197 DAG.getNode(ISD::SUB, DL, VT, Zero, Amt));
1198 }
1199
1200 // Rotate by a immediate.
1201 if (IsCstSplat) {
1202 // ISD::ROTR: Attemp to rotate by a positive immediate.
1203 SDValue Bits = DAG.getConstant(EltSizeInBits, DL, VT);
1204 if (SDValue Urem =
1205 DAG.FoldConstantArithmetic(ISD::UREM, DL, VT, {Amt, Bits}))
1206 return DAG.getNode(Opcode, DL, VT, R, Urem);
1207 }
1208
1209 return Op;
1210}
1211
1212// Return true if Val is equal to (setcc LHS, RHS, CC).
1213// Return false if Val is the inverse of (setcc LHS, RHS, CC).
1214// Otherwise, return std::nullopt.
1215static std::optional<bool> matchSetCC(SDValue LHS, SDValue RHS,
1216 ISD::CondCode CC, SDValue Val) {
1217 assert(Val->getOpcode() == ISD::SETCC);
1218 SDValue LHS2 = Val.getOperand(0);
1219 SDValue RHS2 = Val.getOperand(1);
1220 ISD::CondCode CC2 = cast<CondCodeSDNode>(Val.getOperand(2))->get();
1221
1222 if (LHS == LHS2 && RHS == RHS2) {
1223 if (CC == CC2)
1224 return true;
1225 if (CC == ISD::getSetCCInverse(CC2, LHS2.getValueType()))
1226 return false;
1227 } else if (LHS == RHS2 && RHS == LHS2) {
1229 if (CC == CC2)
1230 return true;
1231 if (CC == ISD::getSetCCInverse(CC2, LHS2.getValueType()))
1232 return false;
1233 }
1234
1235 return std::nullopt;
1236}
1237
1239 const LoongArchSubtarget &Subtarget) {
1240 SDValue CondV = N->getOperand(0);
1241 SDValue TrueV = N->getOperand(1);
1242 SDValue FalseV = N->getOperand(2);
1243 MVT VT = N->getSimpleValueType(0);
1244 SDLoc DL(N);
1245
1246 // (select c, -1, y) -> -c | y
1247 if (isAllOnesConstant(TrueV)) {
1248 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1249 return DAG.getNode(ISD::OR, DL, VT, Neg, DAG.getFreeze(FalseV));
1250 }
1251 // (select c, y, -1) -> (c-1) | y
1252 if (isAllOnesConstant(FalseV)) {
1253 SDValue Neg =
1254 DAG.getNode(ISD::ADD, DL, VT, CondV, DAG.getAllOnesConstant(DL, VT));
1255 return DAG.getNode(ISD::OR, DL, VT, Neg, DAG.getFreeze(TrueV));
1256 }
1257
1258 // (select c, 0, y) -> (c-1) & y
1259 if (isNullConstant(TrueV)) {
1260 SDValue Neg =
1261 DAG.getNode(ISD::ADD, DL, VT, CondV, DAG.getAllOnesConstant(DL, VT));
1262 return DAG.getNode(ISD::AND, DL, VT, Neg, DAG.getFreeze(FalseV));
1263 }
1264 // (select c, y, 0) -> -c & y
1265 if (isNullConstant(FalseV)) {
1266 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1267 return DAG.getNode(ISD::AND, DL, VT, Neg, DAG.getFreeze(TrueV));
1268 }
1269
1270 // select c, ~x, x --> xor -c, x
1271 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV)) {
1272 const APInt &TrueVal = TrueV->getAsAPIntVal();
1273 const APInt &FalseVal = FalseV->getAsAPIntVal();
1274 if (~TrueVal == FalseVal) {
1275 SDValue Neg = DAG.getNegative(CondV, DL, VT);
1276 return DAG.getNode(ISD::XOR, DL, VT, Neg, FalseV);
1277 }
1278 }
1279
1280 // Try to fold (select (setcc lhs, rhs, cc), truev, falsev) into bitwise ops
1281 // when both truev and falsev are also setcc.
1282 if (CondV.getOpcode() == ISD::SETCC && TrueV.getOpcode() == ISD::SETCC &&
1283 FalseV.getOpcode() == ISD::SETCC) {
1284 SDValue LHS = CondV.getOperand(0);
1285 SDValue RHS = CondV.getOperand(1);
1286 ISD::CondCode CC = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1287
1288 // (select x, x, y) -> x | y
1289 // (select !x, x, y) -> x & y
1290 if (std::optional<bool> MatchResult = matchSetCC(LHS, RHS, CC, TrueV)) {
1291 return DAG.getNode(*MatchResult ? ISD::OR : ISD::AND, DL, VT, TrueV,
1292 DAG.getFreeze(FalseV));
1293 }
1294 // (select x, y, x) -> x & y
1295 // (select !x, y, x) -> x | y
1296 if (std::optional<bool> MatchResult = matchSetCC(LHS, RHS, CC, FalseV)) {
1297 return DAG.getNode(*MatchResult ? ISD::AND : ISD::OR, DL, VT,
1298 DAG.getFreeze(TrueV), FalseV);
1299 }
1300 }
1301
1302 return SDValue();
1303}
1304
1305// Transform `binOp (select cond, x, c0), c1` where `c0` and `c1` are constants
1306// into `select cond, binOp(x, c1), binOp(c0, c1)` if profitable.
1307// For now we only consider transformation profitable if `binOp(c0, c1)` ends up
1308// being `0` or `-1`. In such cases we can replace `select` with `and`.
1309// TODO: Should we also do this if `binOp(c0, c1)` is cheaper to materialize
1310// than `c0`?
1311static SDValue
1313 const LoongArchSubtarget &Subtarget) {
1314 unsigned SelOpNo = 0;
1315 SDValue Sel = BO->getOperand(0);
1316 if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse()) {
1317 SelOpNo = 1;
1318 Sel = BO->getOperand(1);
1319 }
1320
1321 if (Sel.getOpcode() != ISD::SELECT || !Sel.hasOneUse())
1322 return SDValue();
1323
1324 unsigned ConstSelOpNo = 1;
1325 unsigned OtherSelOpNo = 2;
1326 if (!isa<ConstantSDNode>(Sel->getOperand(ConstSelOpNo))) {
1327 ConstSelOpNo = 2;
1328 OtherSelOpNo = 1;
1329 }
1330 SDValue ConstSelOp = Sel->getOperand(ConstSelOpNo);
1331 ConstantSDNode *ConstSelOpNode = dyn_cast<ConstantSDNode>(ConstSelOp);
1332 if (!ConstSelOpNode || ConstSelOpNode->isOpaque())
1333 return SDValue();
1334
1335 SDValue ConstBinOp = BO->getOperand(SelOpNo ^ 1);
1336 ConstantSDNode *ConstBinOpNode = dyn_cast<ConstantSDNode>(ConstBinOp);
1337 if (!ConstBinOpNode || ConstBinOpNode->isOpaque())
1338 return SDValue();
1339
1340 SDLoc DL(Sel);
1341 EVT VT = BO->getValueType(0);
1342
1343 SDValue NewConstOps[2] = {ConstSelOp, ConstBinOp};
1344 if (SelOpNo == 1)
1345 std::swap(NewConstOps[0], NewConstOps[1]);
1346
1347 SDValue NewConstOp =
1348 DAG.FoldConstantArithmetic(BO->getOpcode(), DL, VT, NewConstOps);
1349 if (!NewConstOp)
1350 return SDValue();
1351
1352 const APInt &NewConstAPInt = NewConstOp->getAsAPIntVal();
1353 if (!NewConstAPInt.isZero() && !NewConstAPInt.isAllOnes())
1354 return SDValue();
1355
1356 SDValue OtherSelOp = Sel->getOperand(OtherSelOpNo);
1357 SDValue NewNonConstOps[2] = {OtherSelOp, ConstBinOp};
1358 if (SelOpNo == 1)
1359 std::swap(NewNonConstOps[0], NewNonConstOps[1]);
1360 SDValue NewNonConstOp = DAG.getNode(BO->getOpcode(), DL, VT, NewNonConstOps);
1361
1362 SDValue NewT = (ConstSelOpNo == 1) ? NewConstOp : NewNonConstOp;
1363 SDValue NewF = (ConstSelOpNo == 1) ? NewNonConstOp : NewConstOp;
1364 return DAG.getSelect(DL, VT, Sel.getOperand(0), NewT, NewF);
1365}
1366
1367// Changes the condition code and swaps operands if necessary, so the SetCC
1368// operation matches one of the comparisons supported directly by branches
1369// in the LoongArch ISA. May adjust compares to favor compare with 0 over
1370// compare with 1/-1.
1372 ISD::CondCode &CC, SelectionDAG &DAG) {
1373 // If this is a single bit test that can't be handled by ANDI, shift the
1374 // bit to be tested to the MSB and perform a signed compare with 0.
1375 if (isIntEqualitySetCC(CC) && isNullConstant(RHS) &&
1376 LHS.getOpcode() == ISD::AND && LHS.hasOneUse() &&
1377 isa<ConstantSDNode>(LHS.getOperand(1))) {
1378 uint64_t Mask = LHS.getConstantOperandVal(1);
1379 if ((isPowerOf2_64(Mask) || isMask_64(Mask)) && !isInt<12>(Mask)) {
1380 unsigned ShAmt = 0;
1381 if (isPowerOf2_64(Mask)) {
1382 CC = CC == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
1383 ShAmt = LHS.getValueSizeInBits() - 1 - Log2_64(Mask);
1384 } else {
1385 ShAmt = LHS.getValueSizeInBits() - llvm::bit_width(Mask);
1386 }
1387
1388 LHS = LHS.getOperand(0);
1389 if (ShAmt != 0)
1390 LHS = DAG.getNode(ISD::SHL, DL, LHS.getValueType(), LHS,
1391 DAG.getConstant(ShAmt, DL, LHS.getValueType()));
1392 return;
1393 }
1394 }
1395
1396 if (auto *RHSC = dyn_cast<ConstantSDNode>(RHS)) {
1397 int64_t C = RHSC->getSExtValue();
1398 switch (CC) {
1399 default:
1400 break;
1401 case ISD::SETGT:
1402 // Convert X > -1 to X >= 0.
1403 if (C == -1) {
1404 RHS = DAG.getConstant(0, DL, RHS.getValueType());
1405 CC = ISD::SETGE;
1406 return;
1407 }
1408 break;
1409 case ISD::SETLT:
1410 // Convert X < 1 to 0 >= X.
1411 if (C == 1) {
1412 RHS = LHS;
1413 LHS = DAG.getConstant(0, DL, RHS.getValueType());
1414 CC = ISD::SETGE;
1415 return;
1416 }
1417 break;
1418 }
1419 }
1420
1421 switch (CC) {
1422 default:
1423 break;
1424 case ISD::SETGT:
1425 case ISD::SETLE:
1426 case ISD::SETUGT:
1427 case ISD::SETULE:
1429 std::swap(LHS, RHS);
1430 break;
1431 }
1432}
1433
1434SDValue LoongArchTargetLowering::lowerSELECT(SDValue Op,
1435 SelectionDAG &DAG) const {
1436 SDValue CondV = Op.getOperand(0);
1437 SDValue TrueV = Op.getOperand(1);
1438 SDValue FalseV = Op.getOperand(2);
1439 SDLoc DL(Op);
1440 MVT VT = Op.getSimpleValueType();
1441 MVT GRLenVT = Subtarget.getGRLenVT();
1442
1443 if (SDValue V = combineSelectToBinOp(Op.getNode(), DAG, Subtarget))
1444 return V;
1445
1446 if (Op.hasOneUse()) {
1447 unsigned UseOpc = Op->user_begin()->getOpcode();
1448 if (isBinOp(UseOpc) && DAG.isSafeToSpeculativelyExecute(UseOpc)) {
1449 SDNode *BinOp = *Op->user_begin();
1450 if (SDValue NewSel = foldBinOpIntoSelectIfProfitable(*Op->user_begin(),
1451 DAG, Subtarget)) {
1452 DAG.ReplaceAllUsesWith(BinOp, &NewSel);
1453 // Opcode check is necessary because foldBinOpIntoSelectIfProfitable
1454 // may return a constant node and cause crash in lowerSELECT.
1455 if (NewSel.getOpcode() == ISD::SELECT)
1456 return lowerSELECT(NewSel, DAG);
1457 return NewSel;
1458 }
1459 }
1460 }
1461
1462 // If the condition is not an integer SETCC which operates on GRLenVT, we need
1463 // to emit a LoongArchISD::SELECT_CC comparing the condition to zero. i.e.:
1464 // (select condv, truev, falsev)
1465 // -> (loongarchisd::select_cc condv, zero, setne, truev, falsev)
1466 if (CondV.getOpcode() != ISD::SETCC ||
1467 CondV.getOperand(0).getSimpleValueType() != GRLenVT) {
1468 SDValue Zero = DAG.getConstant(0, DL, GRLenVT);
1469 SDValue SetNE = DAG.getCondCode(ISD::SETNE);
1470
1471 SDValue Ops[] = {CondV, Zero, SetNE, TrueV, FalseV};
1472
1473 return DAG.getNode(LoongArchISD::SELECT_CC, DL, VT, Ops);
1474 }
1475
1476 // If the CondV is the output of a SETCC node which operates on GRLenVT
1477 // inputs, then merge the SETCC node into the lowered LoongArchISD::SELECT_CC
1478 // to take advantage of the integer compare+branch instructions. i.e.: (select
1479 // (setcc lhs, rhs, cc), truev, falsev)
1480 // -> (loongarchisd::select_cc lhs, rhs, cc, truev, falsev)
1481 SDValue LHS = CondV.getOperand(0);
1482 SDValue RHS = CondV.getOperand(1);
1483 ISD::CondCode CCVal = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1484
1485 // Special case for a select of 2 constants that have a difference of 1.
1486 // Normally this is done by DAGCombine, but if the select is introduced by
1487 // type legalization or op legalization, we miss it. Restricting to SETLT
1488 // case for now because that is what signed saturating add/sub need.
1489 // FIXME: We don't need the condition to be SETLT or even a SETCC,
1490 // but we would probably want to swap the true/false values if the condition
1491 // is SETGE/SETLE to avoid an XORI.
1492 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV) &&
1493 CCVal == ISD::SETLT) {
1494 const APInt &TrueVal = TrueV->getAsAPIntVal();
1495 const APInt &FalseVal = FalseV->getAsAPIntVal();
1496 if (TrueVal - 1 == FalseVal)
1497 return DAG.getNode(ISD::ADD, DL, VT, CondV, FalseV);
1498 if (TrueVal + 1 == FalseVal)
1499 return DAG.getNode(ISD::SUB, DL, VT, FalseV, CondV);
1500 }
1501
1502 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
1503 // 1 < x ? x : 1 -> 0 < x ? x : 1
1504 if (isOneConstant(LHS) && (CCVal == ISD::SETLT || CCVal == ISD::SETULT) &&
1505 RHS == TrueV && LHS == FalseV) {
1506 LHS = DAG.getConstant(0, DL, VT);
1507 // 0 <u x is the same as x != 0.
1508 if (CCVal == ISD::SETULT) {
1509 std::swap(LHS, RHS);
1510 CCVal = ISD::SETNE;
1511 }
1512 }
1513
1514 // x <s -1 ? x : -1 -> x <s 0 ? x : -1
1515 if (isAllOnesConstant(RHS) && CCVal == ISD::SETLT && LHS == TrueV &&
1516 RHS == FalseV) {
1517 RHS = DAG.getConstant(0, DL, VT);
1518 }
1519
1520 SDValue TargetCC = DAG.getCondCode(CCVal);
1521
1522 if (isa<ConstantSDNode>(TrueV) && !isa<ConstantSDNode>(FalseV)) {
1523 // (select (setcc lhs, rhs, CC), constant, falsev)
1524 // -> (select (setcc lhs, rhs, InverseCC), falsev, constant)
1525 std::swap(TrueV, FalseV);
1526 TargetCC = DAG.getCondCode(ISD::getSetCCInverse(CCVal, LHS.getValueType()));
1527 }
1528
1529 SDValue Ops[] = {LHS, RHS, TargetCC, TrueV, FalseV};
1530 return DAG.getNode(LoongArchISD::SELECT_CC, DL, VT, Ops);
1531}
1532
1533SDValue LoongArchTargetLowering::lowerBRCOND(SDValue Op,
1534 SelectionDAG &DAG) const {
1535 SDValue CondV = Op.getOperand(1);
1536 SDLoc DL(Op);
1537 MVT GRLenVT = Subtarget.getGRLenVT();
1538
1539 if (CondV.getOpcode() == ISD::SETCC) {
1540 if (CondV.getOperand(0).getValueType() == GRLenVT) {
1541 SDValue LHS = CondV.getOperand(0);
1542 SDValue RHS = CondV.getOperand(1);
1543 ISD::CondCode CCVal = cast<CondCodeSDNode>(CondV.getOperand(2))->get();
1544
1545 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
1546
1547 SDValue TargetCC = DAG.getCondCode(CCVal);
1548 return DAG.getNode(LoongArchISD::BR_CC, DL, Op.getValueType(),
1549 Op.getOperand(0), LHS, RHS, TargetCC,
1550 Op.getOperand(2));
1551 } else if (CondV.getOperand(0).getValueType().isFloatingPoint()) {
1552 return DAG.getNode(LoongArchISD::BRCOND, DL, Op.getValueType(),
1553 Op.getOperand(0), CondV, Op.getOperand(2));
1554 }
1555 }
1556
1557 return DAG.getNode(LoongArchISD::BR_CC, DL, Op.getValueType(),
1558 Op.getOperand(0), CondV, DAG.getConstant(0, DL, GRLenVT),
1559 DAG.getCondCode(ISD::SETNE), Op.getOperand(2));
1560}
1561
1562SDValue
1563LoongArchTargetLowering::lowerSCALAR_TO_VECTOR(SDValue Op,
1564 SelectionDAG &DAG) const {
1565 SDLoc DL(Op);
1566 MVT OpVT = Op.getSimpleValueType();
1567
1568 SDValue Vector = DAG.getUNDEF(OpVT);
1569 SDValue Val = Op.getOperand(0);
1570 SDValue Idx = DAG.getConstant(0, DL, Subtarget.getGRLenVT());
1571
1572 return DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, OpVT, Vector, Val, Idx);
1573}
1574
1575SDValue LoongArchTargetLowering::lowerBITREVERSE(SDValue Op,
1576 SelectionDAG &DAG) const {
1577 EVT ResTy = Op->getValueType(0);
1578 SDValue Src = Op->getOperand(0);
1579 SDLoc DL(Op);
1580
1581 // LoongArchISD::BITREV_8B is not supported on LA32.
1582 if (!Subtarget.is64Bit() && (ResTy == MVT::v16i8 || ResTy == MVT::v32i8))
1583 return SDValue();
1584
1585 EVT NewVT = ResTy.is128BitVector() ? MVT::v2i64 : MVT::v4i64;
1586 unsigned int OrigEltNum = ResTy.getVectorNumElements();
1587 unsigned int NewEltNum = NewVT.getVectorNumElements();
1588
1589 SDValue NewSrc = DAG.getNode(ISD::BITCAST, DL, NewVT, Src);
1590
1592 for (unsigned int i = 0; i < NewEltNum; i++) {
1593 SDValue Op = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i64, NewSrc,
1594 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
1595 unsigned RevOp = (ResTy == MVT::v16i8 || ResTy == MVT::v32i8)
1596 ? (unsigned)LoongArchISD::BITREV_8B
1597 : (unsigned)ISD::BITREVERSE;
1598 Ops.push_back(DAG.getNode(RevOp, DL, MVT::i64, Op));
1599 }
1600 SDValue Res =
1601 DAG.getNode(ISD::BITCAST, DL, ResTy, DAG.getBuildVector(NewVT, DL, Ops));
1602
1603 switch (ResTy.getSimpleVT().SimpleTy) {
1604 default:
1605 return SDValue();
1606 case MVT::v16i8:
1607 case MVT::v32i8:
1608 return Res;
1609 case MVT::v8i16:
1610 case MVT::v16i16:
1611 case MVT::v4i32:
1612 case MVT::v8i32: {
1614 for (unsigned int i = 0; i < NewEltNum; i++)
1615 for (int j = OrigEltNum / NewEltNum - 1; j >= 0; j--)
1616 Mask.push_back(j + (OrigEltNum / NewEltNum) * i);
1617 return DAG.getVectorShuffle(ResTy, DL, Res, DAG.getUNDEF(ResTy), Mask);
1618 }
1619 }
1620}
1621
1622// Widen element type to get a new mask value (if possible).
1623// For example:
1624// shufflevector <4 x i32> %a, <4 x i32> %b,
1625// <4 x i32> <i32 6, i32 7, i32 2, i32 3>
1626// is equivalent to:
1627// shufflevector <2 x i64> %a, <2 x i64> %b, <2 x i32> <i32 3, i32 1>
1628// can be lowered to:
1629// VPACKOD_D vr0, vr0, vr1
1631 SDValue V1, SDValue V2, SelectionDAG &DAG) {
1632 unsigned EltBits = VT.getScalarSizeInBits();
1633
1634 if (EltBits > 32 || EltBits == 1)
1635 return SDValue();
1636
1637 SmallVector<int, 8> NewMask;
1638 if (widenShuffleMaskElts(Mask, NewMask)) {
1639 MVT NewEltVT = VT.isFloatingPoint() ? MVT::getFloatingPointVT(EltBits * 2)
1640 : MVT::getIntegerVT(EltBits * 2);
1641 MVT NewVT = MVT::getVectorVT(NewEltVT, VT.getVectorNumElements() / 2);
1642 if (DAG.getTargetLoweringInfo().isTypeLegal(NewVT)) {
1643 SDValue NewV1 = DAG.getBitcast(NewVT, V1);
1644 SDValue NewV2 = DAG.getBitcast(NewVT, V2);
1645 return DAG.getBitcast(
1646 VT, DAG.getVectorShuffle(NewVT, DL, NewV1, NewV2, NewMask));
1647 }
1648 }
1649
1650 return SDValue();
1651}
1652
1653/// Attempts to match a shuffle mask against the VBSLL, VBSRL, VSLLI and VSRLI
1654/// instruction.
1655// The funciton matches elements from one of the input vector shuffled to the
1656// left or right with zeroable elements 'shifted in'. It handles both the
1657// strictly bit-wise element shifts and the byte shfit across an entire 128-bit
1658// lane.
1659// Mostly copied from X86.
1660static int matchShuffleAsShift(MVT &ShiftVT, unsigned &Opcode,
1661 unsigned ScalarSizeInBits, ArrayRef<int> Mask,
1662 int MaskOffset, const APInt &Zeroable) {
1663 int Size = Mask.size();
1664 unsigned SizeInBits = Size * ScalarSizeInBits;
1665
1666 auto CheckZeros = [&](int Shift, int Scale, bool Left) {
1667 for (int i = 0; i < Size; i += Scale)
1668 for (int j = 0; j < Shift; ++j)
1669 if (!Zeroable[i + j + (Left ? 0 : (Scale - Shift))])
1670 return false;
1671
1672 return true;
1673 };
1674
1675 auto isSequentialOrUndefInRange = [&](unsigned Pos, unsigned Size, int Low,
1676 int Step = 1) {
1677 for (unsigned i = Pos, e = Pos + Size; i != e; ++i, Low += Step)
1678 if (!(Mask[i] == -1 || Mask[i] == Low))
1679 return false;
1680 return true;
1681 };
1682
1683 auto MatchShift = [&](int Shift, int Scale, bool Left) {
1684 for (int i = 0; i != Size; i += Scale) {
1685 unsigned Pos = Left ? i + Shift : i;
1686 unsigned Low = Left ? i : i + Shift;
1687 unsigned Len = Scale - Shift;
1688 if (!isSequentialOrUndefInRange(Pos, Len, Low + MaskOffset))
1689 return -1;
1690 }
1691
1692 int ShiftEltBits = ScalarSizeInBits * Scale;
1693 bool ByteShift = ShiftEltBits > 64;
1694 Opcode = Left ? (ByteShift ? LoongArchISD::VBSLL : LoongArchISD::VSLLI)
1695 : (ByteShift ? LoongArchISD::VBSRL : LoongArchISD::VSRLI);
1696 int ShiftAmt = Shift * ScalarSizeInBits / (ByteShift ? 8 : 1);
1697
1698 // Normalize the scale for byte shifts to still produce an i64 element
1699 // type.
1700 Scale = ByteShift ? Scale / 2 : Scale;
1701
1702 // We need to round trip through the appropriate type for the shift.
1703 MVT ShiftSVT = MVT::getIntegerVT(ScalarSizeInBits * Scale);
1704 ShiftVT = ByteShift ? MVT::getVectorVT(MVT::i8, SizeInBits / 8)
1705 : MVT::getVectorVT(ShiftSVT, Size / Scale);
1706 return (int)ShiftAmt;
1707 };
1708
1709 unsigned MaxWidth = 128;
1710 for (int Scale = 2; Scale * ScalarSizeInBits <= MaxWidth; Scale *= 2)
1711 for (int Shift = 1; Shift != Scale; ++Shift)
1712 for (bool Left : {true, false})
1713 if (CheckZeros(Shift, Scale, Left)) {
1714 int ShiftAmt = MatchShift(Shift, Scale, Left);
1715 if (0 < ShiftAmt)
1716 return ShiftAmt;
1717 }
1718
1719 // no match
1720 return -1;
1721}
1722
1723/// Lower VECTOR_SHUFFLE as shift (if possible).
1724///
1725/// For example:
1726/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1727/// <4 x i32> <i32 4, i32 0, i32 1, i32 2>
1728/// is lowered to:
1729/// (VBSLL_V $v0, $v0, 4)
1730///
1731/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1732/// <4 x i32> <i32 4, i32 0, i32 4, i32 2>
1733/// is lowered to:
1734/// (VSLLI_D $v0, $v0, 32)
1736 MVT VT, SDValue V1, SDValue V2,
1737 SelectionDAG &DAG,
1738 const LoongArchSubtarget &Subtarget,
1739 const APInt &Zeroable) {
1740 int Size = Mask.size();
1741 assert(Size == (int)VT.getVectorNumElements() && "Unexpected mask size");
1742
1743 MVT ShiftVT;
1744 SDValue V = V1;
1745 unsigned Opcode;
1746
1747 // Try to match shuffle against V1 shift.
1748 int ShiftAmt = matchShuffleAsShift(ShiftVT, Opcode, VT.getScalarSizeInBits(),
1749 Mask, 0, Zeroable);
1750
1751 // If V1 failed, try to match shuffle against V2 shift.
1752 if (ShiftAmt < 0) {
1753 ShiftAmt = matchShuffleAsShift(ShiftVT, Opcode, VT.getScalarSizeInBits(),
1754 Mask, Size, Zeroable);
1755 V = V2;
1756 }
1757
1758 if (ShiftAmt < 0)
1759 return SDValue();
1760
1761 assert(DAG.getTargetLoweringInfo().isTypeLegal(ShiftVT) &&
1762 "Illegal integer vector type");
1763 V = DAG.getBitcast(ShiftVT, V);
1764 V = DAG.getNode(Opcode, DL, ShiftVT, V,
1765 DAG.getConstant(ShiftAmt, DL, Subtarget.getGRLenVT()));
1766 return DAG.getBitcast(VT, V);
1767}
1768
1769/// Determine whether a range fits a regular pattern of values.
1770/// This function accounts for the possibility of jumping over the End iterator.
1771template <typename ValType>
1772static bool
1774 unsigned CheckStride,
1776 ValType ExpectedIndex, unsigned ExpectedIndexStride) {
1777 auto &I = Begin;
1778
1779 while (I != End) {
1780 if (*I != -1 && *I != ExpectedIndex)
1781 return false;
1782 ExpectedIndex += ExpectedIndexStride;
1783
1784 // Incrementing past End is undefined behaviour so we must increment one
1785 // step at a time and check for End at each step.
1786 for (unsigned n = 0; n < CheckStride && I != End; ++n, ++I)
1787 ; // Empty loop body.
1788 }
1789 return true;
1790}
1791
1792/// Compute whether each element of a shuffle is zeroable.
1793///
1794/// A "zeroable" vector shuffle element is one which can be lowered to zero.
1796 SDValue V2, APInt &KnownUndef,
1797 APInt &KnownZero) {
1798 int Size = Mask.size();
1799 KnownUndef = KnownZero = APInt::getZero(Size);
1800
1802 V2 = peekThroughBitcasts(V2);
1803
1804 bool V1IsZero = ISD::isBuildVectorAllZeros(V1.getNode());
1805 bool V2IsZero = ISD::isBuildVectorAllZeros(V2.getNode());
1806
1807 int VectorSizeInBits = V1.getValueSizeInBits();
1808 int ScalarSizeInBits = VectorSizeInBits / Size;
1809 assert(!(VectorSizeInBits % ScalarSizeInBits) && "Illegal shuffle mask size");
1810 (void)ScalarSizeInBits;
1811
1812 for (int i = 0; i < Size; ++i) {
1813 int M = Mask[i];
1814 if (M < 0) {
1815 KnownUndef.setBit(i);
1816 continue;
1817 }
1818 if ((M >= 0 && M < Size && V1IsZero) || (M >= Size && V2IsZero)) {
1819 KnownZero.setBit(i);
1820 continue;
1821 }
1822 }
1823}
1824
1825/// Test whether a shuffle mask is equivalent within each sub-lane.
1826///
1827/// The specific repeated shuffle mask is populated in \p RepeatedMask, as it is
1828/// non-trivial to compute in the face of undef lanes. The representation is
1829/// suitable for use with existing 128-bit shuffles as entries from the second
1830/// vector have been remapped to [LaneSize, 2*LaneSize).
1831static bool isRepeatedShuffleMask(unsigned LaneSizeInBits, MVT VT,
1832 ArrayRef<int> Mask,
1833 SmallVectorImpl<int> &RepeatedMask) {
1834 auto LaneSize = LaneSizeInBits / VT.getScalarSizeInBits();
1835 RepeatedMask.assign(LaneSize, -1);
1836 int Size = Mask.size();
1837 for (int i = 0; i < Size; ++i) {
1838 assert(Mask[i] == -1 || Mask[i] >= 0);
1839 if (Mask[i] < 0)
1840 continue;
1841 if ((Mask[i] % Size) / LaneSize != i / LaneSize)
1842 // This entry crosses lanes, so there is no way to model this shuffle.
1843 return false;
1844
1845 // Ok, handle the in-lane shuffles by detecting if and when they repeat.
1846 // Adjust second vector indices to start at LaneSize instead of Size.
1847 int LocalM =
1848 Mask[i] < Size ? Mask[i] % LaneSize : Mask[i] % LaneSize + LaneSize;
1849 if (RepeatedMask[i % LaneSize] < 0)
1850 // This is the first non-undef entry in this slot of a 128-bit lane.
1851 RepeatedMask[i % LaneSize] = LocalM;
1852 else if (RepeatedMask[i % LaneSize] != LocalM)
1853 // Found a mismatch with the repeated mask.
1854 return false;
1855 }
1856 return true;
1857}
1858
1859/// Attempts to match vector shuffle as byte rotation.
1861 ArrayRef<int> Mask) {
1862
1863 SDValue Lo, Hi;
1864 SmallVector<int, 16> RepeatedMask;
1865
1866 if (!isRepeatedShuffleMask(128, VT, Mask, RepeatedMask))
1867 return -1;
1868
1869 int NumElts = RepeatedMask.size();
1870 int Rotation = 0;
1871 int Scale = 16 / NumElts;
1872
1873 for (int i = 0; i < NumElts; ++i) {
1874 int M = RepeatedMask[i];
1875 assert((M == -1 || (0 <= M && M < (2 * NumElts))) &&
1876 "Unexpected mask index.");
1877 if (M < 0)
1878 continue;
1879
1880 // Determine where a rotated vector would have started.
1881 int StartIdx = i - (M % NumElts);
1882 if (StartIdx == 0)
1883 return -1;
1884
1885 // If we found the tail of a vector the rotation must be the missing
1886 // front. If we found the head of a vector, it must be how much of the
1887 // head.
1888 int CandidateRotation = StartIdx < 0 ? -StartIdx : NumElts - StartIdx;
1889
1890 if (Rotation == 0)
1891 Rotation = CandidateRotation;
1892 else if (Rotation != CandidateRotation)
1893 return -1;
1894
1895 // Compute which value this mask is pointing at.
1896 SDValue MaskV = M < NumElts ? V1 : V2;
1897
1898 // Compute which of the two target values this index should be assigned
1899 // to. This reflects whether the high elements are remaining or the low
1900 // elements are remaining.
1901 SDValue &TargetV = StartIdx < 0 ? Hi : Lo;
1902
1903 // Either set up this value if we've not encountered it before, or check
1904 // that it remains consistent.
1905 if (!TargetV)
1906 TargetV = MaskV;
1907 else if (TargetV != MaskV)
1908 return -1;
1909 }
1910
1911 // Check that we successfully analyzed the mask, and normalize the results.
1912 assert(Rotation != 0 && "Failed to locate a viable rotation!");
1913 assert((Lo || Hi) && "Failed to find a rotated input vector!");
1914 if (!Lo)
1915 Lo = Hi;
1916 else if (!Hi)
1917 Hi = Lo;
1918
1919 V1 = Lo;
1920 V2 = Hi;
1921
1922 return Rotation * Scale;
1923}
1924
1925/// Lower VECTOR_SHUFFLE as byte rotate (if possible).
1926///
1927/// For example:
1928/// %shuffle = shufflevector <2 x i64> %a, <2 x i64> %b,
1929/// <2 x i32> <i32 3, i32 0>
1930/// is lowered to:
1931/// (VBSRL_V $v1, $v1, 8)
1932/// (VBSLL_V $v0, $v0, 8)
1933/// (VOR_V $v0, $V0, $v1)
1934static SDValue
1936 SDValue V1, SDValue V2, SelectionDAG &DAG,
1937 const LoongArchSubtarget &Subtarget) {
1938
1939 SDValue Lo = V1, Hi = V2;
1940 int ByteRotation = matchShuffleAsByteRotate(VT, Lo, Hi, Mask);
1941 if (ByteRotation <= 0)
1942 return SDValue();
1943
1944 MVT ByteVT = MVT::getVectorVT(MVT::i8, VT.getSizeInBits() / 8);
1945 Lo = DAG.getBitcast(ByteVT, Lo);
1946 Hi = DAG.getBitcast(ByteVT, Hi);
1947
1948 int LoByteShift = 16 - ByteRotation;
1949 int HiByteShift = ByteRotation;
1950 MVT GRLenVT = Subtarget.getGRLenVT();
1951
1952 SDValue LoShift = DAG.getNode(LoongArchISD::VBSLL, DL, ByteVT, Lo,
1953 DAG.getConstant(LoByteShift, DL, GRLenVT));
1954 SDValue HiShift = DAG.getNode(LoongArchISD::VBSRL, DL, ByteVT, Hi,
1955 DAG.getConstant(HiByteShift, DL, GRLenVT));
1956 return DAG.getBitcast(VT, DAG.getNode(ISD::OR, DL, ByteVT, LoShift, HiShift));
1957}
1958
1959/// Lower VECTOR_SHUFFLE as ZERO_EXTEND Or ANY_EXTEND (if possible).
1960///
1961/// For example:
1962/// %2 = shufflevector <4 x i32> %0, <4 x i32> zeroinitializer,
1963/// <4 x i32> <i32 0, i32 4, i32 1, i32 4>
1964/// %3 = bitcast <4 x i32> %2 to <2 x i64>
1965/// is lowered to:
1966/// (VREPLI $v1, 0)
1967/// (VILVL $v0, $v1, $v0)
1969 ArrayRef<int> Mask, MVT VT,
1970 SDValue V1, SDValue V2,
1971 SelectionDAG &DAG,
1972 const APInt &Zeroable) {
1973 int Bits = VT.getSizeInBits();
1974 int EltBits = VT.getScalarSizeInBits();
1975 int NumElements = VT.getVectorNumElements();
1976
1977 if (Zeroable.isAllOnes())
1978 return DAG.getConstant(0, DL, VT);
1979
1980 // Define a helper function to check a particular ext-scale and lower to it if
1981 // valid.
1982 auto Lower = [&](int Scale) -> SDValue {
1983 SDValue InputV;
1984 bool AnyExt = true;
1985 int Offset = 0;
1986 for (int i = 0; i < NumElements; i++) {
1987 int M = Mask[i];
1988 if (M < 0)
1989 continue;
1990 if (i % Scale != 0) {
1991 // Each of the extended elements need to be zeroable.
1992 if (!Zeroable[i])
1993 return SDValue();
1994
1995 AnyExt = false;
1996 continue;
1997 }
1998
1999 // Each of the base elements needs to be consecutive indices into the
2000 // same input vector.
2001 SDValue V = M < NumElements ? V1 : V2;
2002 M = M % NumElements;
2003 if (!InputV) {
2004 InputV = V;
2005 Offset = M - (i / Scale);
2006
2007 // These offset can't be handled
2008 if (Offset % (NumElements / Scale))
2009 return SDValue();
2010 } else if (InputV != V)
2011 return SDValue();
2012
2013 if (M != (Offset + (i / Scale)))
2014 return SDValue(); // Non-consecutive strided elements.
2015 }
2016
2017 // If we fail to find an input, we have a zero-shuffle which should always
2018 // have already been handled.
2019 if (!InputV)
2020 return SDValue();
2021
2022 do {
2023 unsigned VilVLoHi = LoongArchISD::VILVL;
2024 if (Offset >= (NumElements / 2)) {
2025 VilVLoHi = LoongArchISD::VILVH;
2026 Offset -= (NumElements / 2);
2027 }
2028
2029 MVT InputVT = MVT::getVectorVT(MVT::getIntegerVT(EltBits), NumElements);
2030 SDValue Ext =
2031 AnyExt ? DAG.getFreeze(InputV) : DAG.getConstant(0, DL, InputVT);
2032 InputV = DAG.getBitcast(InputVT, InputV);
2033 InputV = DAG.getNode(VilVLoHi, DL, InputVT, Ext, InputV);
2034 Scale /= 2;
2035 EltBits *= 2;
2036 NumElements /= 2;
2037 } while (Scale > 1);
2038 return DAG.getBitcast(VT, InputV);
2039 };
2040
2041 // Each iteration, try extending the elements half as much, but into twice as
2042 // many elements.
2043 for (int NumExtElements = Bits / 64; NumExtElements < NumElements;
2044 NumExtElements *= 2) {
2045 if (SDValue V = Lower(NumElements / NumExtElements))
2046 return V;
2047 }
2048 return SDValue();
2049}
2050
2051/// Lower VECTOR_SHUFFLE into VREPLVEI (if possible).
2052///
2053/// VREPLVEI performs vector broadcast based on an element specified by an
2054/// integer immediate, with its mask being similar to:
2055/// <x, x, x, ...>
2056/// where x is any valid index.
2057///
2058/// When undef's appear in the mask they are treated as if they were whatever
2059/// value is necessary in order to fit the above form.
2060static SDValue
2062 SDValue V1, SelectionDAG &DAG,
2063 const LoongArchSubtarget &Subtarget) {
2064 int SplatIndex = -1;
2065 for (const auto &M : Mask) {
2066 if (M != -1) {
2067 SplatIndex = M;
2068 break;
2069 }
2070 }
2071
2072 if (SplatIndex == -1)
2073 return DAG.getUNDEF(VT);
2074
2075 assert(SplatIndex < (int)Mask.size() && "Out of bounds mask index");
2076 if (fitsRegularPattern<int>(Mask.begin(), 1, Mask.end(), SplatIndex, 0)) {
2077 return DAG.getNode(LoongArchISD::VREPLVEI, DL, VT, V1,
2078 DAG.getConstant(SplatIndex, DL, Subtarget.getGRLenVT()));
2079 }
2080
2081 return SDValue();
2082}
2083
2084/// Lower VECTOR_SHUFFLE into VSHUF4I (if possible).
2085///
2086/// VSHUF4I splits the vector into blocks of four elements, then shuffles these
2087/// elements according to a <4 x i2> constant (encoded as an integer immediate).
2088///
2089/// It is therefore possible to lower into VSHUF4I when the mask takes the form:
2090/// <a, b, c, d, a+4, b+4, c+4, d+4, a+8, b+8, c+8, d+8, ...>
2091/// When undef's appear they are treated as if they were whatever value is
2092/// necessary in order to fit the above forms.
2093///
2094/// For example:
2095/// %2 = shufflevector <8 x i16> %0, <8 x i16> undef,
2096/// <8 x i32> <i32 3, i32 2, i32 1, i32 0,
2097/// i32 7, i32 6, i32 5, i32 4>
2098/// is lowered to:
2099/// (VSHUF4I_H $v0, $v1, 27)
2100/// where the 27 comes from:
2101/// 3 + (2 << 2) + (1 << 4) + (0 << 6)
2102static SDValue
2104 SDValue V1, SDValue V2, SelectionDAG &DAG,
2105 const LoongArchSubtarget &Subtarget) {
2106
2107 unsigned SubVecSize = 4;
2108 if (VT == MVT::v2f64 || VT == MVT::v2i64)
2109 SubVecSize = 2;
2110
2111 int SubMask[4] = {-1, -1, -1, -1};
2112 for (unsigned i = 0; i < SubVecSize; ++i) {
2113 for (unsigned j = i; j < Mask.size(); j += SubVecSize) {
2114 int M = Mask[j];
2115
2116 // Convert from vector index to 4-element subvector index
2117 // If an index refers to an element outside of the subvector then give up
2118 if (M != -1) {
2119 M -= 4 * (j / SubVecSize);
2120 if (M < 0 || M >= 4)
2121 return SDValue();
2122 }
2123
2124 // If the mask has an undef, replace it with the current index.
2125 // Note that it might still be undef if the current index is also undef
2126 if (SubMask[i] == -1)
2127 SubMask[i] = M;
2128 // Check that non-undef values are the same as in the mask. If they
2129 // aren't then give up
2130 else if (M != -1 && M != SubMask[i])
2131 return SDValue();
2132 }
2133 }
2134
2135 // Calculate the immediate. Replace any remaining undefs with zero
2136 int Imm = 0;
2137 for (int i = SubVecSize - 1; i >= 0; --i) {
2138 int M = SubMask[i];
2139
2140 if (M == -1)
2141 M = 0;
2142
2143 Imm <<= 2;
2144 Imm |= M & 0x3;
2145 }
2146
2147 MVT GRLenVT = Subtarget.getGRLenVT();
2148
2149 // Return vshuf4i.d
2150 if (VT == MVT::v2f64 || VT == MVT::v2i64)
2151 return DAG.getNode(LoongArchISD::VSHUF4I_D, DL, VT, V1, V2,
2152 DAG.getConstant(Imm, DL, GRLenVT));
2153
2154 return DAG.getNode(LoongArchISD::VSHUF4I, DL, VT, V1,
2155 DAG.getConstant(Imm, DL, GRLenVT));
2156}
2157
2158/// Lower VECTOR_SHUFFLE whose result is the reversed source vector.
2159///
2160/// It is possible to do optimization for VECTOR_SHUFFLE performing vector
2161/// reverse whose mask likes:
2162/// <7, 6, 5, 4, 3, 2, 1, 0>
2163///
2164/// When undef's appear in the mask they are treated as if they were whatever
2165/// value is necessary in order to fit the above forms.
2166static SDValue
2168 SDValue V1, SelectionDAG &DAG,
2169 const LoongArchSubtarget &Subtarget) {
2170 // Only vectors with i8/i16 elements which cannot match other patterns
2171 // directly needs to do this.
2172 if (VT != MVT::v16i8 && VT != MVT::v8i16 && VT != MVT::v32i8 &&
2173 VT != MVT::v16i16)
2174 return SDValue();
2175
2176 if (!ShuffleVectorInst::isReverseMask(Mask, Mask.size()))
2177 return SDValue();
2178
2179 int WidenNumElts = VT.getVectorNumElements() / 4;
2180 SmallVector<int, 16> WidenMask(WidenNumElts, -1);
2181 for (int i = 0; i < WidenNumElts; ++i)
2182 WidenMask[i] = WidenNumElts - 1 - i;
2183
2184 MVT WidenVT = MVT::getVectorVT(
2185 VT.getVectorElementType() == MVT::i8 ? MVT::i32 : MVT::i64, WidenNumElts);
2186 SDValue NewV1 = DAG.getBitcast(WidenVT, V1);
2187 SDValue WidenRev = DAG.getVectorShuffle(WidenVT, DL, NewV1,
2188 DAG.getUNDEF(WidenVT), WidenMask);
2189
2190 return DAG.getNode(LoongArchISD::VSHUF4I, DL, VT,
2191 DAG.getBitcast(VT, WidenRev),
2192 DAG.getConstant(27, DL, Subtarget.getGRLenVT()));
2193}
2194
2195/// Lower VECTOR_SHUFFLE into VPACKEV (if possible).
2196///
2197/// VPACKEV interleaves the even elements from each vector.
2198///
2199/// It is possible to lower into VPACKEV when the mask consists of two of the
2200/// following forms interleaved:
2201/// <0, 2, 4, ...>
2202/// <n, n+2, n+4, ...>
2203/// where n is the number of elements in the vector.
2204/// For example:
2205/// <0, 0, 2, 2, 4, 4, ...>
2206/// <0, n, 2, n+2, 4, n+4, ...>
2207///
2208/// When undef's appear in the mask they are treated as if they were whatever
2209/// value is necessary in order to fit the above forms.
2211 MVT VT, SDValue V1, SDValue V2,
2212 SelectionDAG &DAG) {
2213
2214 const auto &Begin = Mask.begin();
2215 const auto &End = Mask.end();
2216 SDValue OriV1 = V1, OriV2 = V2;
2217
2218 if (fitsRegularPattern<int>(Begin, 2, End, 0, 2))
2219 V1 = OriV1;
2220 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size(), 2))
2221 V1 = OriV2;
2222 else
2223 return SDValue();
2224
2225 if (fitsRegularPattern<int>(Begin + 1, 2, End, 0, 2))
2226 V2 = OriV1;
2227 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size(), 2))
2228 V2 = OriV2;
2229 else
2230 return SDValue();
2231
2232 return DAG.getNode(LoongArchISD::VPACKEV, DL, VT, V2, V1);
2233}
2234
2235/// Lower VECTOR_SHUFFLE into VPACKOD (if possible).
2236///
2237/// VPACKOD interleaves the odd elements from each vector.
2238///
2239/// It is possible to lower into VPACKOD when the mask consists of two of the
2240/// following forms interleaved:
2241/// <1, 3, 5, ...>
2242/// <n+1, n+3, n+5, ...>
2243/// where n is the number of elements in the vector.
2244/// For example:
2245/// <1, 1, 3, 3, 5, 5, ...>
2246/// <1, n+1, 3, n+3, 5, n+5, ...>
2247///
2248/// When undef's appear in the mask they are treated as if they were whatever
2249/// value is necessary in order to fit the above forms.
2251 MVT VT, SDValue V1, SDValue V2,
2252 SelectionDAG &DAG) {
2253
2254 const auto &Begin = Mask.begin();
2255 const auto &End = Mask.end();
2256 SDValue OriV1 = V1, OriV2 = V2;
2257
2258 if (fitsRegularPattern<int>(Begin, 2, End, 1, 2))
2259 V1 = OriV1;
2260 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size() + 1, 2))
2261 V1 = OriV2;
2262 else
2263 return SDValue();
2264
2265 if (fitsRegularPattern<int>(Begin + 1, 2, End, 1, 2))
2266 V2 = OriV1;
2267 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size() + 1, 2))
2268 V2 = OriV2;
2269 else
2270 return SDValue();
2271
2272 return DAG.getNode(LoongArchISD::VPACKOD, DL, VT, V2, V1);
2273}
2274
2275/// Lower VECTOR_SHUFFLE into VILVH (if possible).
2276///
2277/// VILVH interleaves consecutive elements from the left (highest-indexed) half
2278/// of each vector.
2279///
2280/// It is possible to lower into VILVH when the mask consists of two of the
2281/// following forms interleaved:
2282/// <x, x+1, x+2, ...>
2283/// <n+x, n+x+1, n+x+2, ...>
2284/// where n is the number of elements in the vector and x is half n.
2285/// For example:
2286/// <x, x, x+1, x+1, x+2, x+2, ...>
2287/// <x, n+x, x+1, n+x+1, x+2, n+x+2, ...>
2288///
2289/// When undef's appear in the mask they are treated as if they were whatever
2290/// value is necessary in order to fit the above forms.
2292 MVT VT, SDValue V1, SDValue V2,
2293 SelectionDAG &DAG) {
2294
2295 const auto &Begin = Mask.begin();
2296 const auto &End = Mask.end();
2297 unsigned HalfSize = Mask.size() / 2;
2298 SDValue OriV1 = V1, OriV2 = V2;
2299
2300 if (fitsRegularPattern<int>(Begin, 2, End, HalfSize, 1))
2301 V1 = OriV1;
2302 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size() + HalfSize, 1))
2303 V1 = OriV2;
2304 else
2305 return SDValue();
2306
2307 if (fitsRegularPattern<int>(Begin + 1, 2, End, HalfSize, 1))
2308 V2 = OriV1;
2309 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size() + HalfSize,
2310 1))
2311 V2 = OriV2;
2312 else
2313 return SDValue();
2314
2315 return DAG.getNode(LoongArchISD::VILVH, DL, VT, V2, V1);
2316}
2317
2318/// Lower VECTOR_SHUFFLE into VILVL (if possible).
2319///
2320/// VILVL interleaves consecutive elements from the right (lowest-indexed) half
2321/// of each vector.
2322///
2323/// It is possible to lower into VILVL when the mask consists of two of the
2324/// following forms interleaved:
2325/// <0, 1, 2, ...>
2326/// <n, n+1, n+2, ...>
2327/// where n is the number of elements in the vector.
2328/// For example:
2329/// <0, 0, 1, 1, 2, 2, ...>
2330/// <0, n, 1, n+1, 2, n+2, ...>
2331///
2332/// When undef's appear in the mask they are treated as if they were whatever
2333/// value is necessary in order to fit the above forms.
2335 MVT VT, SDValue V1, SDValue V2,
2336 SelectionDAG &DAG) {
2337
2338 const auto &Begin = Mask.begin();
2339 const auto &End = Mask.end();
2340 SDValue OriV1 = V1, OriV2 = V2;
2341
2342 if (fitsRegularPattern<int>(Begin, 2, End, 0, 1))
2343 V1 = OriV1;
2344 else if (fitsRegularPattern<int>(Begin, 2, End, Mask.size(), 1))
2345 V1 = OriV2;
2346 else
2347 return SDValue();
2348
2349 if (fitsRegularPattern<int>(Begin + 1, 2, End, 0, 1))
2350 V2 = OriV1;
2351 else if (fitsRegularPattern<int>(Begin + 1, 2, End, Mask.size(), 1))
2352 V2 = OriV2;
2353 else
2354 return SDValue();
2355
2356 return DAG.getNode(LoongArchISD::VILVL, DL, VT, V2, V1);
2357}
2358
2359/// Lower VECTOR_SHUFFLE into VPICKEV (if possible).
2360///
2361/// VPICKEV copies the even elements of each vector into the result vector.
2362///
2363/// It is possible to lower into VPICKEV when the mask consists of two of the
2364/// following forms concatenated:
2365/// <0, 2, 4, ...>
2366/// <n, n+2, n+4, ...>
2367/// where n is the number of elements in the vector.
2368/// For example:
2369/// <0, 2, 4, ..., 0, 2, 4, ...>
2370/// <0, 2, 4, ..., n, n+2, n+4, ...>
2371///
2372/// When undef's appear in the mask they are treated as if they were whatever
2373/// value is necessary in order to fit the above forms.
2375 MVT VT, SDValue V1, SDValue V2,
2376 SelectionDAG &DAG) {
2377
2378 const auto &Begin = Mask.begin();
2379 const auto &Mid = Mask.begin() + Mask.size() / 2;
2380 const auto &End = Mask.end();
2381 SDValue OriV1 = V1, OriV2 = V2;
2382
2383 if (fitsRegularPattern<int>(Begin, 1, Mid, 0, 2))
2384 V1 = OriV1;
2385 else if (fitsRegularPattern<int>(Begin, 1, Mid, Mask.size(), 2))
2386 V1 = OriV2;
2387 else
2388 return SDValue();
2389
2390 if (fitsRegularPattern<int>(Mid, 1, End, 0, 2))
2391 V2 = OriV1;
2392 else if (fitsRegularPattern<int>(Mid, 1, End, Mask.size(), 2))
2393 V2 = OriV2;
2394
2395 else
2396 return SDValue();
2397
2398 return DAG.getNode(LoongArchISD::VPICKEV, DL, VT, V2, V1);
2399}
2400
2401/// Lower VECTOR_SHUFFLE into VPICKOD (if possible).
2402///
2403/// VPICKOD copies the odd elements of each vector into the result vector.
2404///
2405/// It is possible to lower into VPICKOD when the mask consists of two of the
2406/// following forms concatenated:
2407/// <1, 3, 5, ...>
2408/// <n+1, n+3, n+5, ...>
2409/// where n is the number of elements in the vector.
2410/// For example:
2411/// <1, 3, 5, ..., 1, 3, 5, ...>
2412/// <1, 3, 5, ..., n+1, n+3, n+5, ...>
2413///
2414/// When undef's appear in the mask they are treated as if they were whatever
2415/// value is necessary in order to fit the above forms.
2417 MVT VT, SDValue V1, SDValue V2,
2418 SelectionDAG &DAG) {
2419
2420 const auto &Begin = Mask.begin();
2421 const auto &Mid = Mask.begin() + Mask.size() / 2;
2422 const auto &End = Mask.end();
2423 SDValue OriV1 = V1, OriV2 = V2;
2424
2425 if (fitsRegularPattern<int>(Begin, 1, Mid, 1, 2))
2426 V1 = OriV1;
2427 else if (fitsRegularPattern<int>(Begin, 1, Mid, Mask.size() + 1, 2))
2428 V1 = OriV2;
2429 else
2430 return SDValue();
2431
2432 if (fitsRegularPattern<int>(Mid, 1, End, 1, 2))
2433 V2 = OriV1;
2434 else if (fitsRegularPattern<int>(Mid, 1, End, Mask.size() + 1, 2))
2435 V2 = OriV2;
2436 else
2437 return SDValue();
2438
2439 return DAG.getNode(LoongArchISD::VPICKOD, DL, VT, V2, V1);
2440}
2441
2442/// Lower VECTOR_SHUFFLE into VEXTRINS (if possible).
2443///
2444/// VEXTRINS copies one element of a vector into any place of the result
2445/// vector and makes no change to the rest elements of the result vector.
2446///
2447/// It is possible to lower into VEXTRINS when the mask takes the form:
2448/// <0, 1, 2, ..., n+i, ..., n-1> or <n, n+1, n+2, ..., i, ..., 2n-1> or
2449/// <0, 1, 2, ..., i, ..., n-1> or <n, n+1, n+2, ..., n+i, ..., 2n-1>
2450/// where n is the number of elements in the vector and i is in [0, n).
2451/// For example:
2452/// <0, 1, 2, 3, 4, 5, 6, 8> , <2, 9, 10, 11, 12, 13, 14, 15> ,
2453/// <0, 1, 2, 6, 4, 5, 6, 7> , <8, 9, 10, 11, 12, 9, 14, 15>
2454///
2455/// When undef's appear in the mask they are treated as if they were whatever
2456/// value is necessary in order to fit the above forms.
2457static SDValue
2459 SDValue V1, SDValue V2, SelectionDAG &DAG,
2460 const LoongArchSubtarget &Subtarget) {
2461 unsigned NumElts = VT.getVectorNumElements();
2462 MVT EltVT = VT.getVectorElementType();
2463 MVT GRLenVT = Subtarget.getGRLenVT();
2464
2465 if (Mask.size() != NumElts)
2466 return SDValue();
2467
2468 auto tryLowerToExtrAndIns = [&](unsigned Base) -> SDValue {
2469 int DiffCount = 0;
2470 int DiffPos = -1;
2471 for (unsigned i = 0; i < NumElts; ++i) {
2472 if (Mask[i] == -1)
2473 continue;
2474 if (Mask[i] != int(Base + i)) {
2475 ++DiffCount;
2476 DiffPos = int(i);
2477 if (DiffCount > 1)
2478 return SDValue();
2479 }
2480 }
2481
2482 // Need exactly one differing element to lower into VEXTRINS.
2483 if (DiffCount != 1)
2484 return SDValue();
2485
2486 // DiffMask must be in [0, 2N).
2487 int DiffMask = Mask[DiffPos];
2488 if (DiffMask < 0 || DiffMask >= int(2 * NumElts))
2489 return SDValue();
2490
2491 // Determine source vector and source index.
2492 SDValue SrcVec;
2493 unsigned SrcIdx;
2494 if (unsigned(DiffMask) < NumElts) {
2495 SrcVec = V1;
2496 SrcIdx = unsigned(DiffMask);
2497 } else {
2498 SrcVec = V2;
2499 SrcIdx = unsigned(DiffMask) - NumElts;
2500 }
2501
2502 // Replace with EXTRACT_VECTOR_ELT + INSERT_VECTOR_ELT, it will match the
2503 // patterns of VEXTRINS in tablegen.
2504 SDValue Extracted = DAG.getNode(
2505 ISD::EXTRACT_VECTOR_ELT, DL, EltVT.isFloatingPoint() ? EltVT : GRLenVT,
2506 SrcVec, DAG.getConstant(SrcIdx, DL, GRLenVT));
2507 SDValue Result =
2508 DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, (Base == 0) ? V1 : V2,
2509 Extracted, DAG.getConstant(DiffPos, DL, GRLenVT));
2510
2511 return Result;
2512 };
2513
2514 // Try [0, n-1) insertion then [n, 2n-1) insertion.
2515 if (SDValue Result = tryLowerToExtrAndIns(0))
2516 return Result;
2517 return tryLowerToExtrAndIns(NumElts);
2518}
2519
2520// Check the Mask and then build SrcVec and MaskImm infos which will
2521// be used to build LoongArchISD nodes for VPERMI_W or XVPERMI_W.
2522// On success, return true. Otherwise, return false.
2525 unsigned &MaskImm) {
2526 unsigned MaskSize = Mask.size();
2527
2528 auto isValid = [&](int M, int Off) {
2529 return (M == -1) || (M >= Off && M < Off + 4);
2530 };
2531
2532 auto buildImm = [&](int MLo, int MHi, unsigned Off, unsigned I) {
2533 auto immPart = [&](int M, unsigned Off) {
2534 return (M == -1 ? 0 : (M - Off)) & 0x3;
2535 };
2536 MaskImm |= immPart(MLo, Off) << (I * 2);
2537 MaskImm |= immPart(MHi, Off) << ((I + 1) * 2);
2538 };
2539
2540 for (unsigned i = 0; i < 4; i += 2) {
2541 int MLo = Mask[i];
2542 int MHi = Mask[i + 1];
2543
2544 if (MaskSize == 8) { // Only v8i32/v8f32 need this check.
2545 auto isValid2 = [&](int &M, int M2) {
2546 // If high half index is undef, it's always valid.
2547 if (M2 == -1)
2548 return true;
2549 if (M == -1) {
2550 // If low half index is undef, use index from high half,
2551 // remapped to low half.
2552 if ((M2 % MaskSize) < 4)
2553 return false;
2554 M = M2 - 4;
2555 return true;
2556 }
2557 // Index in low half must be same as index in high half.
2558 return M2 == M + 4;
2559 };
2560 if (!isValid2(MLo, Mask[i + 4]) || !isValid2(MHi, Mask[i + 5]))
2561 return false;
2562 }
2563
2564 if (isValid(MLo, 0) && isValid(MHi, 0)) {
2565 SrcVec.push_back(V1);
2566 buildImm(MLo, MHi, 0, i);
2567 } else if (isValid(MLo, MaskSize) && isValid(MHi, MaskSize)) {
2568 SrcVec.push_back(V2);
2569 buildImm(MLo, MHi, MaskSize, i);
2570 } else {
2571 return false;
2572 }
2573 }
2574
2575 return true;
2576}
2577
2578/// Lower VECTOR_SHUFFLE into VPERMI (if possible).
2579///
2580/// VPERMI selects two elements from each of the two vectors based on the
2581/// mask and places them in the corresponding positions of the result vector
2582/// in order. Only v4i32 and v4f32 types are allowed.
2583///
2584/// It is possible to lower into VPERMI when the mask consists of two of the
2585/// following forms concatenated:
2586/// <i, j, u, v>
2587/// <u, v, i, j>
2588/// where i,j are in [0,4) and u,v are in [4, 8).
2589/// For example:
2590/// <2, 3, 4, 5>
2591/// <5, 7, 0, 2>
2592///
2593/// When undef's appear in the mask they are treated as if they were whatever
2594/// value is necessary in order to fit the above forms.
2596 MVT VT, SDValue V1, SDValue V2,
2597 SelectionDAG &DAG,
2598 const LoongArchSubtarget &Subtarget) {
2599 if ((VT != MVT::v4i32 && VT != MVT::v4f32) ||
2600 Mask.size() != VT.getVectorNumElements())
2601 return SDValue();
2602
2604 unsigned MaskImm = 0;
2605 if (!buildVPERMIInfo(Mask, V1, V2, SrcVec, MaskImm))
2606 return SDValue();
2607
2608 return DAG.getNode(LoongArchISD::VPERMI, DL, VT, SrcVec[1], SrcVec[0],
2609 DAG.getConstant(MaskImm, DL, Subtarget.getGRLenVT()));
2610}
2611
2612/// Lower VECTOR_SHUFFLE into VSHUF.
2613///
2614/// This mostly consists of converting the shuffle mask into a BUILD_VECTOR and
2615/// adding it as an operand to the resulting VSHUF.
2617 MVT VT, SDValue V1, SDValue V2,
2618 SelectionDAG &DAG,
2619 const LoongArchSubtarget &Subtarget) {
2620
2622 for (auto M : Mask)
2623 Ops.push_back(DAG.getSignedConstant(M, DL, Subtarget.getGRLenVT()));
2624
2625 EVT MaskVecTy = VT.changeVectorElementTypeToInteger();
2626 SDValue MaskVec = DAG.getBuildVector(MaskVecTy, DL, Ops);
2627
2628 // VECTOR_SHUFFLE concatenates the vectors in an vectorwise fashion.
2629 // <0b00, 0b01> + <0b10, 0b11> -> <0b00, 0b01, 0b10, 0b11>
2630 // VSHF concatenates the vectors in a bitwise fashion:
2631 // <0b00, 0b01> + <0b10, 0b11> ->
2632 // 0b0100 + 0b1110 -> 0b01001110
2633 // <0b10, 0b11, 0b00, 0b01>
2634 // We must therefore swap the operands to get the correct result.
2635 return DAG.getNode(LoongArchISD::VSHUF, DL, VT, MaskVec, V2, V1);
2636}
2637
2638/// Dispatching routine to lower various 128-bit LoongArch vector shuffles.
2639///
2640/// This routine breaks down the specific type of 128-bit shuffle and
2641/// dispatches to the lowering routines accordingly.
2643 SDValue V1, SDValue V2, SelectionDAG &DAG,
2644 const LoongArchSubtarget &Subtarget) {
2645 assert((VT.SimpleTy == MVT::v16i8 || VT.SimpleTy == MVT::v8i16 ||
2646 VT.SimpleTy == MVT::v4i32 || VT.SimpleTy == MVT::v2i64 ||
2647 VT.SimpleTy == MVT::v4f32 || VT.SimpleTy == MVT::v2f64) &&
2648 "Vector type is unsupported for lsx!");
2649 assert(V1.getSimpleValueType() == V2.getSimpleValueType() &&
2650 "Two operands have different types!");
2651 assert(VT.getVectorNumElements() == Mask.size() &&
2652 "Unexpected mask size for shuffle!");
2653 assert(Mask.size() % 2 == 0 && "Expected even mask size.");
2654
2655 APInt KnownUndef, KnownZero;
2656 computeZeroableShuffleElements(Mask, V1, V2, KnownUndef, KnownZero);
2657 APInt Zeroable = KnownUndef | KnownZero;
2658
2659 SDValue Result;
2660 // TODO: Add more comparison patterns.
2661 if (V2.isUndef()) {
2662 if ((Result =
2663 lowerVECTOR_SHUFFLE_VREPLVEI(DL, Mask, VT, V1, DAG, Subtarget)))
2664 return Result;
2665 if ((Result =
2666 lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2667 return Result;
2668 if ((Result =
2669 lowerVECTOR_SHUFFLE_IsReverse(DL, Mask, VT, V1, DAG, Subtarget)))
2670 return Result;
2671
2672 // TODO: This comment may be enabled in the future to better match the
2673 // pattern for instruction selection.
2674 /* V2 = V1; */
2675 }
2676
2677 // It is recommended not to change the pattern comparison order for better
2678 // performance.
2679 if ((Result = lowerVECTOR_SHUFFLE_VPACKEV(DL, Mask, VT, V1, V2, DAG)))
2680 return Result;
2681 if ((Result = lowerVECTOR_SHUFFLE_VPACKOD(DL, Mask, VT, V1, V2, DAG)))
2682 return Result;
2683 if ((Result = lowerVECTOR_SHUFFLE_VILVH(DL, Mask, VT, V1, V2, DAG)))
2684 return Result;
2685 if ((Result = lowerVECTOR_SHUFFLE_VILVL(DL, Mask, VT, V1, V2, DAG)))
2686 return Result;
2687 if ((Result = lowerVECTOR_SHUFFLE_VPICKEV(DL, Mask, VT, V1, V2, DAG)))
2688 return Result;
2689 if ((Result = lowerVECTOR_SHUFFLE_VPICKOD(DL, Mask, VT, V1, V2, DAG)))
2690 return Result;
2691 if ((VT.SimpleTy == MVT::v2i64 || VT.SimpleTy == MVT::v2f64) &&
2692 (Result =
2693 lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2694 return Result;
2695 if ((Result =
2696 lowerVECTOR_SHUFFLE_VEXTRINS(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2697 return Result;
2698 if ((Result = lowerVECTOR_SHUFFLEAsShift(DL, Mask, VT, V1, V2, DAG, Subtarget,
2699 Zeroable)))
2700 return Result;
2701 if ((Result =
2702 lowerVECTOR_SHUFFLE_VPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2703 return Result;
2704 if ((Result = lowerVECTOR_SHUFFLEAsZeroOrAnyExtend(DL, Mask, VT, V1, V2, DAG,
2705 Zeroable)))
2706 return Result;
2707 if ((Result = lowerVECTOR_SHUFFLEAsByteRotate(DL, Mask, VT, V1, V2, DAG,
2708 Subtarget)))
2709 return Result;
2710 if (SDValue NewShuffle = widenShuffleMask(DL, Mask, VT, V1, V2, DAG))
2711 return NewShuffle;
2712 if ((Result =
2713 lowerVECTOR_SHUFFLE_VSHUF(DL, Mask, VT, V1, V2, DAG, Subtarget)))
2714 return Result;
2715 return SDValue();
2716}
2717
2718/// Lower VECTOR_SHUFFLE into XVREPLVEI (if possible).
2719///
2720/// It is a XVREPLVEI when the mask is:
2721/// <x, x, x, ..., x+n, x+n, x+n, ...>
2722/// where the number of x is equal to n and n is half the length of vector.
2723///
2724/// When undef's appear in the mask they are treated as if they were whatever
2725/// value is necessary in order to fit the above form.
2726static SDValue
2728 SDValue V1, SelectionDAG &DAG,
2729 const LoongArchSubtarget &Subtarget) {
2730 int SplatIndex = -1;
2731 for (const auto &M : Mask) {
2732 if (M != -1) {
2733 SplatIndex = M;
2734 break;
2735 }
2736 }
2737
2738 if (SplatIndex == -1)
2739 return DAG.getUNDEF(VT);
2740
2741 const auto &Begin = Mask.begin();
2742 const auto &End = Mask.end();
2743 int HalfSize = Mask.size() / 2;
2744
2745 if (SplatIndex >= HalfSize)
2746 return SDValue();
2747
2748 assert(SplatIndex < (int)Mask.size() && "Out of bounds mask index");
2749 if (fitsRegularPattern<int>(Begin, 1, End - HalfSize, SplatIndex, 0) &&
2750 fitsRegularPattern<int>(Begin + HalfSize, 1, End, SplatIndex + HalfSize,
2751 0)) {
2752 return DAG.getNode(LoongArchISD::VREPLVEI, DL, VT, V1,
2753 DAG.getConstant(SplatIndex, DL, Subtarget.getGRLenVT()));
2754 }
2755
2756 return SDValue();
2757}
2758
2759/// Lower VECTOR_SHUFFLE into XVSHUF4I (if possible).
2760static SDValue
2762 SDValue V1, SDValue V2, SelectionDAG &DAG,
2763 const LoongArchSubtarget &Subtarget) {
2764 // XVSHUF4I_D must be handled separately because it is different from other
2765 // types of [X]VSHUF4I instructions.
2766 if (Mask.size() == 4) {
2767 unsigned MaskImm = 0;
2768 for (int i = 1; i >= 0; --i) {
2769 int MLo = Mask[i];
2770 int MHi = Mask[i + 2];
2771 if (!(MLo == -1 || (MLo >= 0 && MLo <= 1) || (MLo >= 4 && MLo <= 5)) ||
2772 !(MHi == -1 || (MHi >= 2 && MHi <= 3) || (MHi >= 6 && MHi <= 7)))
2773 return SDValue();
2774 if (MHi != -1 && MLo != -1 && MHi != MLo + 2)
2775 return SDValue();
2776
2777 MaskImm <<= 2;
2778 if (MLo != -1)
2779 MaskImm |= ((MLo <= 1) ? MLo : (MLo - 2)) & 0x3;
2780 else if (MHi != -1)
2781 MaskImm |= ((MHi <= 3) ? (MHi - 2) : (MHi - 4)) & 0x3;
2782 }
2783
2784 return DAG.getNode(LoongArchISD::VSHUF4I_D, DL, VT, V1, V2,
2785 DAG.getConstant(MaskImm, DL, Subtarget.getGRLenVT()));
2786 }
2787
2788 return lowerVECTOR_SHUFFLE_VSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget);
2789}
2790
2791/// Lower VECTOR_SHUFFLE into XVPERMI (if possible).
2792static SDValue
2794 SDValue V1, SDValue V2, SelectionDAG &DAG,
2795 const LoongArchSubtarget &Subtarget) {
2796 MVT GRLenVT = Subtarget.getGRLenVT();
2797 unsigned MaskSize = Mask.size();
2798 if (MaskSize != VT.getVectorNumElements())
2799 return SDValue();
2800
2801 // Consider XVPERMI_W.
2802 if (VT == MVT::v8i32 || VT == MVT::v8f32) {
2804 unsigned MaskImm = 0;
2805 if (!buildVPERMIInfo(Mask, V1, V2, SrcVec, MaskImm))
2806 return SDValue();
2807
2808 return DAG.getNode(LoongArchISD::VPERMI, DL, VT, SrcVec[1], SrcVec[0],
2809 DAG.getConstant(MaskImm, DL, GRLenVT));
2810 }
2811
2812 // Consider XVPERMI_D.
2813 if (VT == MVT::v4i64 || VT == MVT::v4f64) {
2814 unsigned MaskImm = 0;
2815 for (unsigned i = 0; i < MaskSize; ++i) {
2816 if (Mask[i] == -1)
2817 continue;
2818 if (Mask[i] >= (int)MaskSize)
2819 return SDValue();
2820 MaskImm |= Mask[i] << (i * 2);
2821 }
2822
2823 return DAG.getNode(LoongArchISD::XVPERMI, DL, VT, V1,
2824 DAG.getConstant(MaskImm, DL, GRLenVT));
2825 }
2826
2827 return SDValue();
2828}
2829
2830/// Lower VECTOR_SHUFFLE into XVPERM (if possible).
2832 MVT VT, SDValue V1, SelectionDAG &DAG,
2833 const LoongArchSubtarget &Subtarget) {
2834 // LoongArch LASX only have XVPERM_W.
2835 if (Mask.size() != 8 || (VT != MVT::v8i32 && VT != MVT::v8f32))
2836 return SDValue();
2837
2838 unsigned NumElts = VT.getVectorNumElements();
2839 unsigned HalfSize = NumElts / 2;
2840 bool FrontLo = true, FrontHi = true;
2841 bool BackLo = true, BackHi = true;
2842
2843 auto inRange = [](int val, int low, int high) {
2844 return (val == -1) || (val >= low && val < high);
2845 };
2846
2847 for (unsigned i = 0; i < HalfSize; ++i) {
2848 int Fronti = Mask[i];
2849 int Backi = Mask[i + HalfSize];
2850
2851 FrontLo &= inRange(Fronti, 0, HalfSize);
2852 FrontHi &= inRange(Fronti, HalfSize, NumElts);
2853 BackLo &= inRange(Backi, 0, HalfSize);
2854 BackHi &= inRange(Backi, HalfSize, NumElts);
2855 }
2856
2857 // If both the lower and upper 128-bit parts access only one half of the
2858 // vector (either lower or upper), avoid using xvperm.w. The latency of
2859 // xvperm.w(3) is higher than using xvshuf(1) and xvori(1).
2860 if ((FrontLo || FrontHi) && (BackLo || BackHi))
2861 return SDValue();
2862
2864 MVT GRLenVT = Subtarget.getGRLenVT();
2865 for (unsigned i = 0; i < NumElts; ++i)
2866 Masks.push_back(Mask[i] == -1 ? DAG.getUNDEF(GRLenVT)
2867 : DAG.getConstant(Mask[i], DL, GRLenVT));
2868 SDValue MaskVec = DAG.getBuildVector(MVT::v8i32, DL, Masks);
2869
2870 return DAG.getNode(LoongArchISD::XVPERM, DL, VT, V1, MaskVec);
2871}
2872
2873/// Lower VECTOR_SHUFFLE into XVPACKEV (if possible).
2875 MVT VT, SDValue V1, SDValue V2,
2876 SelectionDAG &DAG) {
2877 return lowerVECTOR_SHUFFLE_VPACKEV(DL, Mask, VT, V1, V2, DAG);
2878}
2879
2880/// Lower VECTOR_SHUFFLE into XVPACKOD (if possible).
2882 MVT VT, SDValue V1, SDValue V2,
2883 SelectionDAG &DAG) {
2884 return lowerVECTOR_SHUFFLE_VPACKOD(DL, Mask, VT, V1, V2, DAG);
2885}
2886
2887/// Lower VECTOR_SHUFFLE into XVILVH (if possible).
2889 MVT VT, SDValue V1, SDValue V2,
2890 SelectionDAG &DAG) {
2891
2892 const auto &Begin = Mask.begin();
2893 const auto &End = Mask.end();
2894 unsigned HalfSize = Mask.size() / 2;
2895 unsigned LeftSize = HalfSize / 2;
2896 SDValue OriV1 = V1, OriV2 = V2;
2897
2898 if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, HalfSize - LeftSize,
2899 1) &&
2900 fitsRegularPattern<int>(Begin + HalfSize, 2, End, HalfSize + LeftSize, 1))
2901 V1 = OriV1;
2902 else if (fitsRegularPattern<int>(Begin, 2, End - HalfSize,
2903 Mask.size() + HalfSize - LeftSize, 1) &&
2904 fitsRegularPattern<int>(Begin + HalfSize, 2, End,
2905 Mask.size() + HalfSize + LeftSize, 1))
2906 V1 = OriV2;
2907 else
2908 return SDValue();
2909
2910 if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, HalfSize - LeftSize,
2911 1) &&
2912 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End, HalfSize + LeftSize,
2913 1))
2914 V2 = OriV1;
2915 else if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize,
2916 Mask.size() + HalfSize - LeftSize, 1) &&
2917 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End,
2918 Mask.size() + HalfSize + LeftSize, 1))
2919 V2 = OriV2;
2920 else
2921 return SDValue();
2922
2923 return DAG.getNode(LoongArchISD::VILVH, DL, VT, V2, V1);
2924}
2925
2926/// Lower VECTOR_SHUFFLE into XVILVL (if possible).
2928 MVT VT, SDValue V1, SDValue V2,
2929 SelectionDAG &DAG) {
2930
2931 const auto &Begin = Mask.begin();
2932 const auto &End = Mask.end();
2933 unsigned HalfSize = Mask.size() / 2;
2934 SDValue OriV1 = V1, OriV2 = V2;
2935
2936 if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, 0, 1) &&
2937 fitsRegularPattern<int>(Begin + HalfSize, 2, End, HalfSize, 1))
2938 V1 = OriV1;
2939 else if (fitsRegularPattern<int>(Begin, 2, End - HalfSize, Mask.size(), 1) &&
2940 fitsRegularPattern<int>(Begin + HalfSize, 2, End,
2941 Mask.size() + HalfSize, 1))
2942 V1 = OriV2;
2943 else
2944 return SDValue();
2945
2946 if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, 0, 1) &&
2947 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End, HalfSize, 1))
2948 V2 = OriV1;
2949 else if (fitsRegularPattern<int>(Begin + 1, 2, End - HalfSize, Mask.size(),
2950 1) &&
2951 fitsRegularPattern<int>(Begin + 1 + HalfSize, 2, End,
2952 Mask.size() + HalfSize, 1))
2953 V2 = OriV2;
2954 else
2955 return SDValue();
2956
2957 return DAG.getNode(LoongArchISD::VILVL, DL, VT, V2, V1);
2958}
2959
2960/// Lower VECTOR_SHUFFLE into XVPICKEV (if possible).
2962 MVT VT, SDValue V1, SDValue V2,
2963 SelectionDAG &DAG) {
2964
2965 const auto &Begin = Mask.begin();
2966 const auto &LeftMid = Mask.begin() + Mask.size() / 4;
2967 const auto &Mid = Mask.begin() + Mask.size() / 2;
2968 const auto &RightMid = Mask.end() - Mask.size() / 4;
2969 const auto &End = Mask.end();
2970 unsigned HalfSize = Mask.size() / 2;
2971 SDValue OriV1 = V1, OriV2 = V2;
2972
2973 if (fitsRegularPattern<int>(Begin, 1, LeftMid, 0, 2) &&
2974 fitsRegularPattern<int>(Mid, 1, RightMid, HalfSize, 2))
2975 V1 = OriV1;
2976 else if (fitsRegularPattern<int>(Begin, 1, LeftMid, Mask.size(), 2) &&
2977 fitsRegularPattern<int>(Mid, 1, RightMid, Mask.size() + HalfSize, 2))
2978 V1 = OriV2;
2979 else
2980 return SDValue();
2981
2982 if (fitsRegularPattern<int>(LeftMid, 1, Mid, 0, 2) &&
2983 fitsRegularPattern<int>(RightMid, 1, End, HalfSize, 2))
2984 V2 = OriV1;
2985 else if (fitsRegularPattern<int>(LeftMid, 1, Mid, Mask.size(), 2) &&
2986 fitsRegularPattern<int>(RightMid, 1, End, Mask.size() + HalfSize, 2))
2987 V2 = OriV2;
2988
2989 else
2990 return SDValue();
2991
2992 return DAG.getNode(LoongArchISD::VPICKEV, DL, VT, V2, V1);
2993}
2994
2995/// Lower VECTOR_SHUFFLE into XVPICKOD (if possible).
2997 MVT VT, SDValue V1, SDValue V2,
2998 SelectionDAG &DAG) {
2999
3000 const auto &Begin = Mask.begin();
3001 const auto &LeftMid = Mask.begin() + Mask.size() / 4;
3002 const auto &Mid = Mask.begin() + Mask.size() / 2;
3003 const auto &RightMid = Mask.end() - Mask.size() / 4;
3004 const auto &End = Mask.end();
3005 unsigned HalfSize = Mask.size() / 2;
3006 SDValue OriV1 = V1, OriV2 = V2;
3007
3008 if (fitsRegularPattern<int>(Begin, 1, LeftMid, 1, 2) &&
3009 fitsRegularPattern<int>(Mid, 1, RightMid, HalfSize + 1, 2))
3010 V1 = OriV1;
3011 else if (fitsRegularPattern<int>(Begin, 1, LeftMid, Mask.size() + 1, 2) &&
3012 fitsRegularPattern<int>(Mid, 1, RightMid, Mask.size() + HalfSize + 1,
3013 2))
3014 V1 = OriV2;
3015 else
3016 return SDValue();
3017
3018 if (fitsRegularPattern<int>(LeftMid, 1, Mid, 1, 2) &&
3019 fitsRegularPattern<int>(RightMid, 1, End, HalfSize + 1, 2))
3020 V2 = OriV1;
3021 else if (fitsRegularPattern<int>(LeftMid, 1, Mid, Mask.size() + 1, 2) &&
3022 fitsRegularPattern<int>(RightMid, 1, End, Mask.size() + HalfSize + 1,
3023 2))
3024 V2 = OriV2;
3025 else
3026 return SDValue();
3027
3028 return DAG.getNode(LoongArchISD::VPICKOD, DL, VT, V2, V1);
3029}
3030
3031/// Lower VECTOR_SHUFFLE into XVEXTRINS (if possible).
3032static SDValue
3034 SDValue V1, SDValue V2, SelectionDAG &DAG,
3035 const LoongArchSubtarget &Subtarget) {
3036 int NumElts = VT.getVectorNumElements();
3037 int HalfSize = NumElts / 2;
3038 MVT EltVT = VT.getVectorElementType();
3039 MVT GRLenVT = Subtarget.getGRLenVT();
3040
3041 if ((int)Mask.size() != NumElts)
3042 return SDValue();
3043
3044 auto tryLowerToExtrAndIns = [&](int Base) -> SDValue {
3045 SmallVector<int> DiffPos;
3046 for (int i = 0; i < NumElts; ++i) {
3047 if (Mask[i] == -1)
3048 continue;
3049 if (Mask[i] != Base + i) {
3050 DiffPos.push_back(i);
3051 if (DiffPos.size() > 2)
3052 return SDValue();
3053 }
3054 }
3055
3056 // Need exactly two differing element to lower into XVEXTRINS.
3057 // If only one differing element, the element at a distance of
3058 // HalfSize from it must be undef.
3059 if (DiffPos.size() == 1) {
3060 if (DiffPos[0] < HalfSize && Mask[DiffPos[0] + HalfSize] == -1)
3061 DiffPos.push_back(DiffPos[0] + HalfSize);
3062 else if (DiffPos[0] >= HalfSize && Mask[DiffPos[0] - HalfSize] == -1)
3063 DiffPos.insert(DiffPos.begin(), DiffPos[0] - HalfSize);
3064 else
3065 return SDValue();
3066 }
3067 if (DiffPos.size() != 2 || DiffPos[1] != DiffPos[0] + HalfSize)
3068 return SDValue();
3069
3070 // DiffMask must be in its low or high part.
3071 int DiffMaskLo = Mask[DiffPos[0]];
3072 int DiffMaskHi = Mask[DiffPos[1]];
3073 DiffMaskLo = DiffMaskLo == -1 ? DiffMaskHi - HalfSize : DiffMaskLo;
3074 DiffMaskHi = DiffMaskHi == -1 ? DiffMaskLo + HalfSize : DiffMaskHi;
3075 if (!(DiffMaskLo >= 0 && DiffMaskLo < HalfSize) &&
3076 !(DiffMaskLo >= NumElts && DiffMaskLo < NumElts + HalfSize))
3077 return SDValue();
3078 if (!(DiffMaskHi >= HalfSize && DiffMaskHi < NumElts) &&
3079 !(DiffMaskHi >= NumElts + HalfSize && DiffMaskHi < 2 * NumElts))
3080 return SDValue();
3081 if (DiffMaskHi != DiffMaskLo + HalfSize)
3082 return SDValue();
3083
3084 // Determine source vector and source index.
3085 SDValue SrcVec = (DiffMaskLo < HalfSize) ? V1 : V2;
3086 int SrcIdxLo =
3087 (DiffMaskLo < HalfSize) ? DiffMaskLo : (DiffMaskLo - NumElts);
3088 bool IsEltFP = EltVT.isFloatingPoint();
3089
3090 // Replace with 2*EXTRACT_VECTOR_ELT + 2*INSERT_VECTOR_ELT, it will match
3091 // the patterns of XVEXTRINS in tablegen.
3092 SDValue BaseVec = (Base == 0) ? V1 : V2;
3093 SDValue EltLo =
3094 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, IsEltFP ? EltVT : GRLenVT,
3095 SrcVec, DAG.getConstant(SrcIdxLo, DL, GRLenVT));
3096 SDValue InsLo = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, BaseVec, EltLo,
3097 DAG.getConstant(DiffPos[0], DL, GRLenVT));
3098 SDValue EltHi =
3099 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, IsEltFP ? EltVT : GRLenVT,
3100 SrcVec, DAG.getConstant(SrcIdxLo + HalfSize, DL, GRLenVT));
3101 SDValue Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, InsLo, EltHi,
3102 DAG.getConstant(DiffPos[1], DL, GRLenVT));
3103
3104 return Result;
3105 };
3106
3107 // Try [0, n-1) insertion then [n, 2n-1) insertion.
3108 if (SDValue Result = tryLowerToExtrAndIns(0))
3109 return Result;
3110 return tryLowerToExtrAndIns(NumElts);
3111}
3112
3113/// Lower VECTOR_SHUFFLE into XVINSVE0 (if possible).
3114static SDValue
3116 SDValue V1, SDValue V2, SelectionDAG &DAG,
3117 const LoongArchSubtarget &Subtarget) {
3118 // LoongArch LASX only supports xvinsve0.{w/d}.
3119 if (VT != MVT::v8i32 && VT != MVT::v8f32 && VT != MVT::v4i64 &&
3120 VT != MVT::v4f64)
3121 return SDValue();
3122
3123 MVT GRLenVT = Subtarget.getGRLenVT();
3124 int MaskSize = Mask.size();
3125 assert(MaskSize == (int)VT.getVectorNumElements() && "Unexpected mask size");
3126
3127 // Check if exactly one element of the Mask is replaced by 'Replaced', while
3128 // all other elements are either 'Base + i' or undef (-1). On success, return
3129 // the index of the replaced element. Otherwise, just return -1.
3130 auto checkReplaceOne = [&](int Base, int Replaced) -> int {
3131 int Idx = -1;
3132 for (int i = 0; i < MaskSize; ++i) {
3133 if (Mask[i] == Base + i || Mask[i] == -1)
3134 continue;
3135 if (Mask[i] != Replaced)
3136 return -1;
3137 if (Idx == -1)
3138 Idx = i;
3139 else
3140 return -1;
3141 }
3142 return Idx;
3143 };
3144
3145 // Case 1: the lowest element of V2 replaces one element in V1.
3146 int Idx = checkReplaceOne(0, MaskSize);
3147 if (Idx != -1)
3148 return DAG.getNode(LoongArchISD::XVINSVE0, DL, VT, V1, V2,
3149 DAG.getConstant(Idx, DL, GRLenVT));
3150
3151 // Case 2: the lowest element of V1 replaces one element in V2.
3152 Idx = checkReplaceOne(MaskSize, 0);
3153 if (Idx != -1)
3154 return DAG.getNode(LoongArchISD::XVINSVE0, DL, VT, V2, V1,
3155 DAG.getConstant(Idx, DL, GRLenVT));
3156
3157 return SDValue();
3158}
3159
3160/// Lower VECTOR_SHUFFLE into XVSHUF (if possible).
3162 MVT VT, SDValue V1, SDValue V2,
3163 SelectionDAG &DAG) {
3164
3165 int MaskSize = Mask.size();
3166 int HalfSize = Mask.size() / 2;
3167 const auto &Begin = Mask.begin();
3168 const auto &Mid = Mask.begin() + HalfSize;
3169 const auto &End = Mask.end();
3170
3171 // VECTOR_SHUFFLE concatenates the vectors:
3172 // <0, 1, 2, 3, 4, 5, 6, 7> + <8, 9, 10, 11, 12, 13, 14, 15>
3173 // shuffling ->
3174 // <0, 1, 2, 3, 8, 9, 10, 11> <4, 5, 6, 7, 12, 13, 14, 15>
3175 //
3176 // XVSHUF concatenates the vectors:
3177 // <a0, a1, a2, a3, b0, b1, b2, b3> + <a4, a5, a6, a7, b4, b5, b6, b7>
3178 // shuffling ->
3179 // <a0, a1, a2, a3, a4, a5, a6, a7> + <b0, b1, b2, b3, b4, b5, b6, b7>
3180 SmallVector<SDValue, 8> MaskAlloc;
3181 for (auto it = Begin; it < Mid; it++) {
3182 if (*it < 0) // UNDEF
3183 MaskAlloc.push_back(DAG.getTargetConstant(0, DL, MVT::i64));
3184 else if ((*it >= 0 && *it < HalfSize) ||
3185 (*it >= MaskSize && *it < MaskSize + HalfSize)) {
3186 int M = *it < HalfSize ? *it : *it - HalfSize;
3187 MaskAlloc.push_back(DAG.getTargetConstant(M, DL, MVT::i64));
3188 } else
3189 return SDValue();
3190 }
3191 assert((int)MaskAlloc.size() == HalfSize && "xvshuf convert failed!");
3192
3193 for (auto it = Mid; it < End; it++) {
3194 if (*it < 0) // UNDEF
3195 MaskAlloc.push_back(DAG.getTargetConstant(0, DL, MVT::i64));
3196 else if ((*it >= HalfSize && *it < MaskSize) ||
3197 (*it >= MaskSize + HalfSize && *it < MaskSize * 2)) {
3198 int M = *it < MaskSize ? *it - HalfSize : *it - MaskSize;
3199 MaskAlloc.push_back(DAG.getTargetConstant(M, DL, MVT::i64));
3200 } else
3201 return SDValue();
3202 }
3203 assert((int)MaskAlloc.size() == MaskSize && "xvshuf convert failed!");
3204
3205 EVT MaskVecTy = VT.changeVectorElementTypeToInteger();
3206 SDValue MaskVec = DAG.getBuildVector(MaskVecTy, DL, MaskAlloc);
3207 return DAG.getNode(LoongArchISD::VSHUF, DL, VT, MaskVec, V2, V1);
3208}
3209
3210/// Shuffle vectors by lane to generate more optimized instructions.
3211/// 256-bit shuffles are always considered as 2-lane 128-bit shuffles.
3212///
3213/// Therefore, except for the following four cases, other cases are regarded
3214/// as cross-lane shuffles, where optimization is relatively limited.
3215///
3216/// - Shuffle high, low lanes of two inputs vector
3217/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <0, 5, 3, 6>
3218/// - Shuffle low, high lanes of two inputs vector
3219/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <3, 6, 0, 5>
3220/// - Shuffle low, low lanes of two inputs vector
3221/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <3, 6, 3, 6>
3222/// - Shuffle high, high lanes of two inputs vector
3223/// <0, 1, 2, 3> + <4, 5, 6, 7> --- <0, 5, 0, 5>
3224///
3225/// The first case is the closest to LoongArch instructions and the other
3226/// cases need to be converted to it for processing.
3227///
3228/// This function will return true for the last three cases above and will
3229/// modify V1, V2 and Mask. Otherwise, return false for the first case and
3230/// cross-lane shuffle cases.
3232 const SDLoc &DL, MutableArrayRef<int> Mask, MVT VT, SDValue &V1,
3233 SDValue &V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget) {
3234
3235 enum HalfMaskType { HighLaneTy, LowLaneTy, None };
3236
3237 int MaskSize = Mask.size();
3238 int HalfSize = Mask.size() / 2;
3239 MVT GRLenVT = Subtarget.getGRLenVT();
3240
3241 HalfMaskType preMask = None, postMask = None;
3242
3243 if (std::all_of(Mask.begin(), Mask.begin() + HalfSize, [&](int M) {
3244 return M < 0 || (M >= 0 && M < HalfSize) ||
3245 (M >= MaskSize && M < MaskSize + HalfSize);
3246 }))
3247 preMask = HighLaneTy;
3248 else if (std::all_of(Mask.begin(), Mask.begin() + HalfSize, [&](int M) {
3249 return M < 0 || (M >= HalfSize && M < MaskSize) ||
3250 (M >= MaskSize + HalfSize && M < MaskSize * 2);
3251 }))
3252 preMask = LowLaneTy;
3253
3254 if (std::all_of(Mask.begin() + HalfSize, Mask.end(), [&](int M) {
3255 return M < 0 || (M >= HalfSize && M < MaskSize) ||
3256 (M >= MaskSize + HalfSize && M < MaskSize * 2);
3257 }))
3258 postMask = LowLaneTy;
3259 else if (std::all_of(Mask.begin() + HalfSize, Mask.end(), [&](int M) {
3260 return M < 0 || (M >= 0 && M < HalfSize) ||
3261 (M >= MaskSize && M < MaskSize + HalfSize);
3262 }))
3263 postMask = HighLaneTy;
3264
3265 // The pre-half of mask is high lane type, and the post-half of mask
3266 // is low lane type, which is closest to the LoongArch instructions.
3267 //
3268 // Note: In the LoongArch architecture, the high lane of mask corresponds
3269 // to the lower 128-bit of vector register, and the low lane of mask
3270 // corresponds the higher 128-bit of vector register.
3271 if (preMask == HighLaneTy && postMask == LowLaneTy) {
3272 return false;
3273 }
3274 if (preMask == LowLaneTy && postMask == HighLaneTy) {
3275 V1 = DAG.getBitcast(MVT::v4i64, V1);
3276 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3277 DAG.getConstant(0b01001110, DL, GRLenVT));
3278 V1 = DAG.getBitcast(VT, V1);
3279
3280 if (!V2.isUndef()) {
3281 V2 = DAG.getBitcast(MVT::v4i64, V2);
3282 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3283 DAG.getConstant(0b01001110, DL, GRLenVT));
3284 V2 = DAG.getBitcast(VT, V2);
3285 }
3286
3287 for (auto it = Mask.begin(); it < Mask.begin() + HalfSize; it++) {
3288 *it = *it < 0 ? *it : *it - HalfSize;
3289 }
3290 for (auto it = Mask.begin() + HalfSize; it < Mask.end(); it++) {
3291 *it = *it < 0 ? *it : *it + HalfSize;
3292 }
3293 } else if (preMask == LowLaneTy && postMask == LowLaneTy) {
3294 V1 = DAG.getBitcast(MVT::v4i64, V1);
3295 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3296 DAG.getConstant(0b11101110, DL, GRLenVT));
3297 V1 = DAG.getBitcast(VT, V1);
3298
3299 if (!V2.isUndef()) {
3300 V2 = DAG.getBitcast(MVT::v4i64, V2);
3301 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3302 DAG.getConstant(0b11101110, DL, GRLenVT));
3303 V2 = DAG.getBitcast(VT, V2);
3304 }
3305
3306 for (auto it = Mask.begin(); it < Mask.begin() + HalfSize; it++) {
3307 *it = *it < 0 ? *it : *it - HalfSize;
3308 }
3309 } else if (preMask == HighLaneTy && postMask == HighLaneTy) {
3310 V1 = DAG.getBitcast(MVT::v4i64, V1);
3311 V1 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V1,
3312 DAG.getConstant(0b01000100, DL, GRLenVT));
3313 V1 = DAG.getBitcast(VT, V1);
3314
3315 if (!V2.isUndef()) {
3316 V2 = DAG.getBitcast(MVT::v4i64, V2);
3317 V2 = DAG.getNode(LoongArchISD::XVPERMI, DL, MVT::v4i64, V2,
3318 DAG.getConstant(0b01000100, DL, GRLenVT));
3319 V2 = DAG.getBitcast(VT, V2);
3320 }
3321
3322 for (auto it = Mask.begin() + HalfSize; it < Mask.end(); it++) {
3323 *it = *it < 0 ? *it : *it + HalfSize;
3324 }
3325 } else { // cross-lane
3326 return false;
3327 }
3328
3329 return true;
3330}
3331
3332/// Lower VECTOR_SHUFFLE as lane permute and then shuffle (if possible).
3333/// Only for 256-bit vector.
3334///
3335/// For example:
3336/// %2 = shufflevector <4 x i64> %0, <4 x i64> posion,
3337/// <4 x i64> <i32 0, i32 3, i32 2, i32 0>
3338/// is lowerded to:
3339/// (XVPERMI $xr2, $xr0, 78)
3340/// (XVSHUF $xr1, $xr2, $xr0)
3341/// (XVORI $xr0, $xr1, 0)
3343 ArrayRef<int> Mask,
3344 MVT VT, SDValue V1,
3345 SDValue V2,
3346 SelectionDAG &DAG) {
3347 assert(VT.is256BitVector() && "Only for 256-bit vector shuffles!");
3348 int Size = Mask.size();
3349 int LaneSize = Size / 2;
3350
3351 bool LaneCrossing[2] = {false, false};
3352 for (int i = 0; i < Size; ++i)
3353 if (Mask[i] >= 0 && ((Mask[i] % Size) / LaneSize) != (i / LaneSize))
3354 LaneCrossing[(Mask[i] % Size) / LaneSize] = true;
3355
3356 // Ensure that all lanes ared involved.
3357 if (!LaneCrossing[0] && !LaneCrossing[1])
3358 return SDValue();
3359
3360 SmallVector<int> InLaneMask;
3361 InLaneMask.assign(Mask.begin(), Mask.end());
3362 for (int i = 0; i < Size; ++i) {
3363 int &M = InLaneMask[i];
3364 if (M < 0)
3365 continue;
3366 if (((M % Size) / LaneSize) != (i / LaneSize))
3367 M = (M % LaneSize) + ((i / LaneSize) * LaneSize) + Size;
3368 }
3369
3370 SDValue Flipped = DAG.getBitcast(MVT::v4i64, V1);
3371 Flipped = DAG.getVectorShuffle(MVT::v4i64, DL, Flipped,
3372 DAG.getUNDEF(MVT::v4i64), {2, 3, 0, 1});
3373 Flipped = DAG.getBitcast(VT, Flipped);
3374 return DAG.getVectorShuffle(VT, DL, V1, Flipped, InLaneMask);
3375}
3376
3377/// Dispatching routine to lower various 256-bit LoongArch vector shuffles.
3378///
3379/// This routine breaks down the specific type of 256-bit shuffle and
3380/// dispatches to the lowering routines accordingly.
3382 SDValue V1, SDValue V2, SelectionDAG &DAG,
3383 const LoongArchSubtarget &Subtarget) {
3384 assert((VT.SimpleTy == MVT::v32i8 || VT.SimpleTy == MVT::v16i16 ||
3385 VT.SimpleTy == MVT::v8i32 || VT.SimpleTy == MVT::v4i64 ||
3386 VT.SimpleTy == MVT::v8f32 || VT.SimpleTy == MVT::v4f64) &&
3387 "Vector type is unsupported for lasx!");
3388 assert(V1.getSimpleValueType() == V2.getSimpleValueType() &&
3389 "Two operands have different types!");
3390 assert(VT.getVectorNumElements() == Mask.size() &&
3391 "Unexpected mask size for shuffle!");
3392 assert(Mask.size() % 2 == 0 && "Expected even mask size.");
3393 assert(Mask.size() >= 4 && "Mask size is less than 4.");
3394
3395 APInt KnownUndef, KnownZero;
3396 computeZeroableShuffleElements(Mask, V1, V2, KnownUndef, KnownZero);
3397 APInt Zeroable = KnownUndef | KnownZero;
3398
3399 SDValue Result;
3400 // TODO: Add more comparison patterns.
3401 if (V2.isUndef()) {
3402 if ((Result =
3403 lowerVECTOR_SHUFFLE_XVREPLVEI(DL, Mask, VT, V1, DAG, Subtarget)))
3404 return Result;
3405 if ((Result = lowerVECTOR_SHUFFLE_XVSHUF4I(DL, Mask, VT, V1, V2, DAG,
3406 Subtarget)))
3407 return Result;
3408 // Try to widen vectors to gain more optimization opportunities.
3409 if (SDValue NewShuffle = widenShuffleMask(DL, Mask, VT, V1, V2, DAG))
3410 return NewShuffle;
3411 if ((Result =
3412 lowerVECTOR_SHUFFLE_XVPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3413 return Result;
3414 if ((Result = lowerVECTOR_SHUFFLE_XVPERM(DL, Mask, VT, V1, DAG, Subtarget)))
3415 return Result;
3416 if ((Result =
3417 lowerVECTOR_SHUFFLE_IsReverse(DL, Mask, VT, V1, DAG, Subtarget)))
3418 return Result;
3419
3420 // TODO: This comment may be enabled in the future to better match the
3421 // pattern for instruction selection.
3422 /* V2 = V1; */
3423 }
3424
3425 // It is recommended not to change the pattern comparison order for better
3426 // performance.
3427 if ((Result = lowerVECTOR_SHUFFLE_XVPACKEV(DL, Mask, VT, V1, V2, DAG)))
3428 return Result;
3429 if ((Result = lowerVECTOR_SHUFFLE_XVPACKOD(DL, Mask, VT, V1, V2, DAG)))
3430 return Result;
3431 if ((Result = lowerVECTOR_SHUFFLE_XVILVH(DL, Mask, VT, V1, V2, DAG)))
3432 return Result;
3433 if ((Result = lowerVECTOR_SHUFFLE_XVILVL(DL, Mask, VT, V1, V2, DAG)))
3434 return Result;
3435 if ((Result = lowerVECTOR_SHUFFLE_XVPICKEV(DL, Mask, VT, V1, V2, DAG)))
3436 return Result;
3437 if ((Result = lowerVECTOR_SHUFFLE_XVPICKOD(DL, Mask, VT, V1, V2, DAG)))
3438 return Result;
3439 if ((VT.SimpleTy == MVT::v4i64 || VT.SimpleTy == MVT::v4f64) &&
3440 (Result =
3441 lowerVECTOR_SHUFFLE_XVSHUF4I(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3442 return Result;
3443 if ((Result =
3444 lowerVECTOR_SHUFFLE_XVEXTRINS(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3445 return Result;
3446 if ((Result = lowerVECTOR_SHUFFLEAsShift(DL, Mask, VT, V1, V2, DAG, Subtarget,
3447 Zeroable)))
3448 return Result;
3449 if ((Result =
3450 lowerVECTOR_SHUFFLE_XVPERMI(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3451 return Result;
3452 if ((Result =
3453 lowerVECTOR_SHUFFLE_XVINSVE0(DL, Mask, VT, V1, V2, DAG, Subtarget)))
3454 return Result;
3455 if ((Result = lowerVECTOR_SHUFFLEAsByteRotate(DL, Mask, VT, V1, V2, DAG,
3456 Subtarget)))
3457 return Result;
3458
3459 // canonicalize non cross-lane shuffle vector
3460 SmallVector<int> NewMask(Mask);
3461 if (canonicalizeShuffleVectorByLane(DL, NewMask, VT, V1, V2, DAG, Subtarget))
3462 return lower256BitShuffle(DL, NewMask, VT, V1, V2, DAG, Subtarget);
3463
3464 // FIXME: Handling the remaining cases earlier can degrade performance
3465 // in some situations. Further analysis is required to enable more
3466 // effective optimizations.
3467 if (V2.isUndef()) {
3468 if ((Result = lowerVECTOR_SHUFFLEAsLanePermuteAndShuffle(DL, NewMask, VT,
3469 V1, V2, DAG)))
3470 return Result;
3471 }
3472
3473 if (SDValue NewShuffle = widenShuffleMask(DL, NewMask, VT, V1, V2, DAG))
3474 return NewShuffle;
3475 if ((Result = lowerVECTOR_SHUFFLE_XVSHUF(DL, NewMask, VT, V1, V2, DAG)))
3476 return Result;
3477
3478 return SDValue();
3479}
3480
3481SDValue LoongArchTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
3482 SelectionDAG &DAG) const {
3483 ShuffleVectorSDNode *SVOp = cast<ShuffleVectorSDNode>(Op);
3484 ArrayRef<int> OrigMask = SVOp->getMask();
3485 SDValue V1 = Op.getOperand(0);
3486 SDValue V2 = Op.getOperand(1);
3487 MVT VT = Op.getSimpleValueType();
3488 int NumElements = VT.getVectorNumElements();
3489 SDLoc DL(Op);
3490
3491 bool V1IsUndef = V1.isUndef();
3492 bool V2IsUndef = V2.isUndef();
3493 if (V1IsUndef && V2IsUndef)
3494 return DAG.getUNDEF(VT);
3495
3496 // When we create a shuffle node we put the UNDEF node to second operand,
3497 // but in some cases the first operand may be transformed to UNDEF.
3498 // In this case we should just commute the node.
3499 if (V1IsUndef)
3500 return DAG.getCommutedVectorShuffle(*SVOp);
3501
3502 // Check for non-undef masks pointing at an undef vector and make the masks
3503 // undef as well. This makes it easier to match the shuffle based solely on
3504 // the mask.
3505 if (V2IsUndef &&
3506 any_of(OrigMask, [NumElements](int M) { return M >= NumElements; })) {
3507 SmallVector<int, 8> NewMask(OrigMask);
3508 for (int &M : NewMask)
3509 if (M >= NumElements)
3510 M = -1;
3511 return DAG.getVectorShuffle(VT, DL, V1, V2, NewMask);
3512 }
3513
3514 // Check for illegal shuffle mask element index values.
3515 int MaskUpperLimit = OrigMask.size() * (V2IsUndef ? 1 : 2);
3516 (void)MaskUpperLimit;
3517 assert(llvm::all_of(OrigMask,
3518 [&](int M) { return -1 <= M && M < MaskUpperLimit; }) &&
3519 "Out of bounds shuffle index");
3520
3521 // For each vector width, delegate to a specialized lowering routine.
3522 if (VT.is128BitVector())
3523 return lower128BitShuffle(DL, OrigMask, VT, V1, V2, DAG, Subtarget);
3524
3525 if (VT.is256BitVector())
3526 return lower256BitShuffle(DL, OrigMask, VT, V1, V2, DAG, Subtarget);
3527
3528 return SDValue();
3529}
3530
3531SDValue LoongArchTargetLowering::lowerFP_TO_FP16(SDValue Op,
3532 SelectionDAG &DAG) const {
3533 // Custom lower to ensure the libcall return is passed in an FPR on hard
3534 // float ABIs.
3535 SDLoc DL(Op);
3536 MakeLibCallOptions CallOptions;
3537 SDValue Op0 = Op.getOperand(0);
3538 SDValue Chain = SDValue();
3539 RTLIB::Libcall LC = RTLIB::getFPROUND(Op0.getValueType(), MVT::f16);
3540 SDValue Res;
3541 std::tie(Res, Chain) =
3542 makeLibCall(DAG, LC, MVT::f32, Op0, CallOptions, DL, Chain);
3543 if (Subtarget.is64Bit())
3544 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Res);
3545 return DAG.getBitcast(MVT::i32, Res);
3546}
3547
3548SDValue LoongArchTargetLowering::lowerFP16_TO_FP(SDValue Op,
3549 SelectionDAG &DAG) const {
3550 // Custom lower to ensure the libcall argument is passed in an FPR on hard
3551 // float ABIs.
3552 SDLoc DL(Op);
3553 MakeLibCallOptions CallOptions;
3554 SDValue Op0 = Op.getOperand(0);
3555 SDValue Chain = SDValue();
3556 SDValue Arg = Subtarget.is64Bit() ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64,
3557 DL, MVT::f32, Op0)
3558 : DAG.getBitcast(MVT::f32, Op0);
3559 SDValue Res;
3560 std::tie(Res, Chain) = makeLibCall(DAG, RTLIB::FPEXT_F16_F32, MVT::f32, Arg,
3561 CallOptions, DL, Chain);
3562 return Res;
3563}
3564
3565SDValue LoongArchTargetLowering::lowerFP_TO_BF16(SDValue Op,
3566 SelectionDAG &DAG) const {
3567 assert(Subtarget.hasBasicF() && "Unexpected custom legalization");
3568 SDLoc DL(Op);
3569 MakeLibCallOptions CallOptions;
3570 RTLIB::Libcall LC =
3571 RTLIB::getFPROUND(Op.getOperand(0).getValueType(), MVT::bf16);
3572 SDValue Res =
3573 makeLibCall(DAG, LC, MVT::f32, Op.getOperand(0), CallOptions, DL).first;
3574 if (Subtarget.is64Bit())
3575 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Res);
3576 return DAG.getBitcast(MVT::i32, Res);
3577}
3578
3579SDValue LoongArchTargetLowering::lowerBF16_TO_FP(SDValue Op,
3580 SelectionDAG &DAG) const {
3581 assert(Subtarget.hasBasicF() && "Unexpected custom legalization");
3582 MVT VT = Op.getSimpleValueType();
3583 SDLoc DL(Op);
3584 Op = DAG.getNode(
3585 ISD::SHL, DL, Op.getOperand(0).getValueType(), Op.getOperand(0),
3586 DAG.getShiftAmountConstant(16, Op.getOperand(0).getValueType(), DL));
3587 SDValue Res = Subtarget.is64Bit() ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64,
3588 DL, MVT::f32, Op)
3589 : DAG.getBitcast(MVT::f32, Op);
3590 if (VT != MVT::f32)
3591 return DAG.getNode(ISD::FP_EXTEND, DL, VT, Res);
3592 return Res;
3593}
3594
3595// Lower BUILD_VECTOR as broadcast load (if possible).
3596// For example:
3597// %a = load i8, ptr %ptr
3598// %b = build_vector %a, %a, %a, %a
3599// is lowered to :
3600// (VLDREPL_B $a0, 0)
3602 const SDLoc &DL,
3603 SelectionDAG &DAG) {
3604 MVT VT = BVOp->getSimpleValueType(0);
3605 int NumOps = BVOp->getNumOperands();
3606
3607 assert((VT.is128BitVector() || VT.is256BitVector()) &&
3608 "Unsupported vector type for broadcast.");
3609
3610 SDValue IdentitySrc;
3611 bool IsIdeneity = true;
3612
3613 for (int i = 0; i != NumOps; i++) {
3614 SDValue Op = BVOp->getOperand(i);
3615 if (Op.getOpcode() != ISD::LOAD || (IdentitySrc && Op != IdentitySrc)) {
3616 IsIdeneity = false;
3617 break;
3618 }
3619 IdentitySrc = BVOp->getOperand(0);
3620 }
3621
3622 // make sure that this load is valid and only has one user.
3623 if (!IsIdeneity || !IdentitySrc || !BVOp->isOnlyUserOf(IdentitySrc.getNode()))
3624 return SDValue();
3625
3626 auto *LN = cast<LoadSDNode>(IdentitySrc);
3627 auto ExtType = LN->getExtensionType();
3628
3629 if ((ExtType == ISD::EXTLOAD || ExtType == ISD::NON_EXTLOAD) &&
3630 VT.getScalarSizeInBits() == LN->getMemoryVT().getScalarSizeInBits()) {
3631 // Indexed loads and stores are not supported on LoongArch.
3632 assert(LN->isUnindexed() && "Unexpected indexed load.");
3633
3634 SDVTList Tys = DAG.getVTList(VT, MVT::Other);
3635 // The offset operand of unindexed load is always undefined, so there is
3636 // no need to pass it to VLDREPL.
3637 SDValue Ops[] = {LN->getChain(), LN->getBasePtr()};
3638 SDValue BCast = DAG.getNode(LoongArchISD::VLDREPL, DL, Tys, Ops);
3639 DAG.ReplaceAllUsesOfValueWith(SDValue(LN, 1), BCast.getValue(1));
3640 return BCast;
3641 }
3642 return SDValue();
3643}
3644
3645// Sequentially insert elements from Ops into Vector, from low to high indices.
3646// Note: Ops can have fewer elements than Vector.
3648 const LoongArchSubtarget &Subtarget, SDValue &Vector,
3649 EVT ResTy) {
3650 assert(Ops.size() <= ResTy.getVectorNumElements());
3651
3652 SDValue Op0 = Ops[0];
3653 if (!Op0.isUndef())
3654 Vector = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, ResTy, Op0);
3655 for (unsigned i = 1; i < Ops.size(); ++i) {
3656 SDValue Opi = Ops[i];
3657 if (Opi.isUndef())
3658 continue;
3659 Vector = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, ResTy, Vector, Opi,
3660 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
3661 }
3662}
3663
3664// Build a ResTy subvector from Node, taking NumElts elements starting at index
3665// 'first'.
3667 SelectionDAG &DAG, SDLoc DL,
3668 const LoongArchSubtarget &Subtarget,
3669 EVT ResTy, unsigned first) {
3670 unsigned NumElts = ResTy.getVectorNumElements();
3671
3672 assert(first + NumElts <= Node->getSimpleValueType(0).getVectorNumElements());
3673
3674 SmallVector<SDValue, 16> Ops(Node->op_begin() + first,
3675 Node->op_begin() + first + NumElts);
3676 SDValue Vector = DAG.getUNDEF(ResTy);
3677 fillVector(Ops, DAG, DL, Subtarget, Vector, ResTy);
3678 return Vector;
3679}
3680
3681SDValue LoongArchTargetLowering::lowerBUILD_VECTOR(SDValue Op,
3682 SelectionDAG &DAG) const {
3683 BuildVectorSDNode *Node = cast<BuildVectorSDNode>(Op);
3684 MVT VT = Node->getSimpleValueType(0);
3685 EVT ResTy = Op->getValueType(0);
3686 unsigned NumElts = ResTy.getVectorNumElements();
3687 SDLoc DL(Op);
3688 APInt SplatValue, SplatUndef;
3689 unsigned SplatBitSize;
3690 bool HasAnyUndefs;
3691 bool IsConstant = false;
3692 bool UseSameConstant = true;
3693 SDValue ConstantValue;
3694 bool Is128Vec = ResTy.is128BitVector();
3695 bool Is256Vec = ResTy.is256BitVector();
3696
3697 if ((!Subtarget.hasExtLSX() || !Is128Vec) &&
3698 (!Subtarget.hasExtLASX() || !Is256Vec))
3699 return SDValue();
3700
3701 if (SDValue Result = lowerBUILD_VECTORAsBroadCastLoad(Node, DL, DAG))
3702 return Result;
3703
3704 if (Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize, HasAnyUndefs,
3705 /*MinSplatBits=*/8) &&
3706 SplatBitSize <= 64) {
3707 // We can only cope with 8, 16, 32, or 64-bit elements.
3708 if (SplatBitSize != 8 && SplatBitSize != 16 && SplatBitSize != 32 &&
3709 SplatBitSize != 64)
3710 return SDValue();
3711
3712 if (SplatBitSize == 64 && !Subtarget.is64Bit()) {
3713 // We can only handle 64-bit elements that are within
3714 // the signed 10-bit range or match vldi patterns on 32-bit targets.
3715 // See the BUILD_VECTOR case in LoongArchDAGToDAGISel::Select().
3716 if (!SplatValue.isSignedIntN(10) &&
3717 !isImmVLDILegalForMode1(SplatValue, SplatBitSize).first)
3718 return SDValue();
3719 if ((Is128Vec && ResTy == MVT::v4i32) ||
3720 (Is256Vec && ResTy == MVT::v8i32))
3721 return Op;
3722 }
3723
3724 EVT ViaVecTy;
3725
3726 switch (SplatBitSize) {
3727 default:
3728 return SDValue();
3729 case 8:
3730 ViaVecTy = Is128Vec ? MVT::v16i8 : MVT::v32i8;
3731 break;
3732 case 16:
3733 ViaVecTy = Is128Vec ? MVT::v8i16 : MVT::v16i16;
3734 break;
3735 case 32:
3736 ViaVecTy = Is128Vec ? MVT::v4i32 : MVT::v8i32;
3737 break;
3738 case 64:
3739 ViaVecTy = Is128Vec ? MVT::v2i64 : MVT::v4i64;
3740 break;
3741 }
3742
3743 // SelectionDAG::getConstant will promote SplatValue appropriately.
3744 SDValue Result = DAG.getConstant(SplatValue, DL, ViaVecTy);
3745
3746 // Bitcast to the type we originally wanted.
3747 if (ViaVecTy != ResTy)
3748 Result = DAG.getNode(ISD::BITCAST, SDLoc(Node), ResTy, Result);
3749
3750 return Result;
3751 }
3752
3753 if (DAG.isSplatValue(Op, /*AllowUndefs=*/false))
3754 return Op;
3755
3756 for (unsigned i = 0; i < NumElts; ++i) {
3757 SDValue Opi = Node->getOperand(i);
3758 if (isIntOrFPConstant(Opi)) {
3759 IsConstant = true;
3760 if (!ConstantValue.getNode())
3761 ConstantValue = Opi;
3762 else if (ConstantValue != Opi)
3763 UseSameConstant = false;
3764 }
3765 }
3766
3767 // If the type of BUILD_VECTOR is v2f64, custom legalizing it has no benefits.
3768 if (IsConstant && UseSameConstant && ResTy != MVT::v2f64) {
3769 SDValue Result = DAG.getSplatBuildVector(ResTy, DL, ConstantValue);
3770 for (unsigned i = 0; i < NumElts; ++i) {
3771 SDValue Opi = Node->getOperand(i);
3772 if (!isIntOrFPConstant(Opi))
3773 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, ResTy, Result, Opi,
3774 DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
3775 }
3776 return Result;
3777 }
3778
3779 if (!IsConstant) {
3780 // If the BUILD_VECTOR has a repeated pattern, use INSERT_VECTOR_ELT to fill
3781 // the sub-sequence of the vector and then broadcast the sub-sequence.
3782 //
3783 // TODO: If the BUILD_VECTOR contains undef elements, consider falling
3784 // back to use INSERT_VECTOR_ELT to materialize the vector, because it
3785 // generates worse code in some cases. This could be further optimized
3786 // with more consideration.
3788 BitVector UndefElements;
3789 if (Node->getRepeatedSequence(Sequence, &UndefElements) &&
3790 UndefElements.count() == 0) {
3791 // Using LSX instructions to fill the sub-sequence of 256-bits vector,
3792 // because the high part can be simply treated as undef.
3793 SDValue Vector = DAG.getUNDEF(ResTy);
3794 EVT FillTy = Is256Vec
3796 : ResTy;
3797 SDValue FillVec =
3798 Is256Vec ? DAG.getExtractSubvector(DL, FillTy, Vector, 0) : Vector;
3799
3800 fillVector(Sequence, DAG, DL, Subtarget, FillVec, FillTy);
3801
3802 unsigned SeqLen = Sequence.size();
3803 unsigned SplatLen = NumElts / SeqLen;
3804 MVT SplatEltTy = MVT::getIntegerVT(VT.getScalarSizeInBits() * SeqLen);
3805 MVT SplatTy = MVT::getVectorVT(SplatEltTy, SplatLen);
3806
3807 // If size of the sub-sequence is half of a 256-bits vector, bitcast the
3808 // vector to v4i64 type in order to match the pattern of XVREPLVE0Q.
3809 if (SplatEltTy == MVT::i128)
3810 SplatTy = MVT::v4i64;
3811
3812 SDValue SplatVec;
3813 SDValue SrcVec = DAG.getBitcast(
3814 SplatTy,
3815 Is256Vec ? DAG.getInsertSubvector(DL, Vector, FillVec, 0) : FillVec);
3816 if (Is256Vec) {
3817 SplatVec =
3818 DAG.getNode((SplatEltTy == MVT::i128) ? LoongArchISD::XVREPLVE0Q
3819 : LoongArchISD::XVREPLVE0,
3820 DL, SplatTy, SrcVec);
3821 } else {
3822 SplatVec = DAG.getNode(LoongArchISD::VREPLVEI, DL, SplatTy, SrcVec,
3823 DAG.getConstant(0, DL, Subtarget.getGRLenVT()));
3824 }
3825
3826 return DAG.getBitcast(ResTy, SplatVec);
3827 }
3828
3829 // Use INSERT_VECTOR_ELT operations rather than expand to stores, because
3830 // using memory operations is much lower.
3831 //
3832 // For 256-bit vectors, normally split into two halves and concatenate.
3833 // Special case: for v8i32/v8f32/v4i64/v4f64, if the upper half has only
3834 // one non-undef element, skip spliting to avoid a worse result.
3835 if (ResTy == MVT::v8i32 || ResTy == MVT::v8f32 || ResTy == MVT::v4i64 ||
3836 ResTy == MVT::v4f64) {
3837 unsigned NonUndefCount = 0;
3838 for (unsigned i = NumElts / 2; i < NumElts; ++i) {
3839 if (!Node->getOperand(i).isUndef()) {
3840 ++NonUndefCount;
3841 if (NonUndefCount > 1)
3842 break;
3843 }
3844 }
3845 if (NonUndefCount == 1)
3846 return fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget, ResTy, 0);
3847 }
3848
3849 EVT VecTy =
3850 Is256Vec ? ResTy.getHalfNumVectorElementsVT(*DAG.getContext()) : ResTy;
3851 SDValue Vector =
3852 fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget, VecTy, 0);
3853
3854 if (Is128Vec)
3855 return Vector;
3856
3857 SDValue VectorHi = fillSubVectorFromBuildVector(Node, DAG, DL, Subtarget,
3858 VecTy, NumElts / 2);
3859
3860 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResTy, Vector, VectorHi);
3861 }
3862
3863 return SDValue();
3864}
3865
3866SDValue LoongArchTargetLowering::lowerCONCAT_VECTORS(SDValue Op,
3867 SelectionDAG &DAG) const {
3868 SDLoc DL(Op);
3869 MVT ResVT = Op.getSimpleValueType();
3870 assert(ResVT.is256BitVector() && Op.getNumOperands() == 2);
3871
3872 if (Op.getOperand(0).getOpcode() == ISD::TRUNCATE &&
3873 Op.getOperand(1).getOpcode() == ISD::TRUNCATE)
3874 return Op;
3875
3876 unsigned NumOperands = Op.getNumOperands();
3877 unsigned NumFreezeUndef = 0;
3878 unsigned NumZero = 0;
3879 unsigned NumNonZero = 0;
3880 unsigned NonZeros = 0;
3881 SmallSet<SDValue, 4> Undefs;
3882 for (unsigned i = 0; i != NumOperands; ++i) {
3883 SDValue SubVec = Op.getOperand(i);
3884 if (SubVec.isUndef())
3885 continue;
3886 if (ISD::isFreezeUndef(SubVec.getNode())) {
3887 // If the freeze(undef) has multiple uses then we must fold to zero.
3888 if (SubVec.hasOneUse()) {
3889 ++NumFreezeUndef;
3890 } else {
3891 ++NumZero;
3892 Undefs.insert(SubVec);
3893 }
3894 } else if (ISD::isBuildVectorAllZeros(SubVec.getNode()))
3895 ++NumZero;
3896 else {
3897 assert(i < sizeof(NonZeros) * CHAR_BIT); // Ensure the shift is in range.
3898 NonZeros |= 1 << i;
3899 ++NumNonZero;
3900 }
3901 }
3902
3903 // If we have more than 2 non-zeros, build each half separately.
3904 if (NumNonZero > 2) {
3905 MVT HalfVT = ResVT.getHalfNumVectorElementsVT();
3906 ArrayRef<SDUse> Ops = Op->ops();
3907 SDValue Lo = DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
3908 Ops.slice(0, NumOperands / 2));
3909 SDValue Hi = DAG.getNode(ISD::CONCAT_VECTORS, DL, HalfVT,
3910 Ops.slice(NumOperands / 2));
3911 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ResVT, Lo, Hi);
3912 }
3913
3914 // Otherwise, build it up through insert_subvectors.
3915 SDValue Vec = NumZero ? DAG.getConstant(0, DL, ResVT)
3916 : (NumFreezeUndef ? DAG.getFreeze(DAG.getUNDEF(ResVT))
3917 : DAG.getUNDEF(ResVT));
3918
3919 // Replace Undef operands with ZeroVector.
3920 for (SDValue U : Undefs)
3921 DAG.ReplaceAllUsesWith(U, DAG.getConstant(0, DL, U.getSimpleValueType()));
3922
3923 MVT SubVT = Op.getOperand(0).getSimpleValueType();
3924 unsigned NumSubElems = SubVT.getVectorNumElements();
3925 for (unsigned i = 0; i != NumOperands; ++i) {
3926 if ((NonZeros & (1 << i)) == 0)
3927 continue;
3928
3929 Vec = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, ResVT, Vec, Op.getOperand(i),
3930 DAG.getVectorIdxConstant(i * NumSubElems, DL));
3931 }
3932
3933 return Vec;
3934}
3935
3936SDValue
3937LoongArchTargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
3938 SelectionDAG &DAG) const {
3939 MVT EltVT = Op.getSimpleValueType();
3940 SDValue Vec = Op->getOperand(0);
3941 EVT VecTy = Vec->getValueType(0);
3942 SDValue Idx = Op->getOperand(1);
3943 SDLoc DL(Op);
3944 MVT GRLenVT = Subtarget.getGRLenVT();
3945
3946 assert(VecTy.is256BitVector() && "Unexpected EXTRACT_VECTOR_ELT vector type");
3947
3948 if (isa<ConstantSDNode>(Idx))
3949 return Op;
3950
3951 switch (VecTy.getSimpleVT().SimpleTy) {
3952 default:
3953 llvm_unreachable("Unexpected type");
3954 case MVT::v32i8:
3955 case MVT::v16i16:
3956 case MVT::v4i64:
3957 case MVT::v4f64: {
3958 // Extract the high half subvector and place it to the low half of a new
3959 // vector. It doesn't matter what the high half of the new vector is.
3960 EVT HalfTy = VecTy.getHalfNumVectorElementsVT(*DAG.getContext());
3961 SDValue VecHi =
3962 DAG.getExtractSubvector(DL, HalfTy, Vec, HalfTy.getVectorNumElements());
3963 SDValue TmpVec =
3964 DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VecTy, DAG.getUNDEF(VecTy),
3965 VecHi, DAG.getConstant(0, DL, GRLenVT));
3966
3967 // Shuffle the origin Vec and the TmpVec using MaskVec, the lowest element
3968 // of MaskVec is Idx, the rest do not matter. ResVec[0] will hold the
3969 // desired element.
3970 SDValue IdxCp =
3971 Subtarget.is64Bit()
3972 ? DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64, DL, MVT::f32, Idx)
3973 : DAG.getBitcast(MVT::f32, Idx);
3974 SDValue IdxVec = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v8f32, IdxCp);
3975 SDValue MaskVec =
3976 DAG.getBitcast((VecTy == MVT::v4f64) ? MVT::v4i64 : VecTy, IdxVec);
3977 SDValue ResVec =
3978 DAG.getNode(LoongArchISD::VSHUF, DL, VecTy, MaskVec, TmpVec, Vec);
3979
3980 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, ResVec,
3981 DAG.getConstant(0, DL, GRLenVT));
3982 }
3983 case MVT::v8i32:
3984 case MVT::v8f32: {
3985 SDValue SplatIdx = DAG.getSplatBuildVector(MVT::v8i32, DL, Idx);
3986 SDValue SplatValue =
3987 DAG.getNode(LoongArchISD::XVPERM, DL, VecTy, Vec, SplatIdx);
3988
3989 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, EltVT, SplatValue,
3990 DAG.getConstant(0, DL, GRLenVT));
3991 }
3992 }
3993}
3994
3995SDValue
3996LoongArchTargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
3997 SelectionDAG &DAG) const {
3998 MVT VT = Op.getSimpleValueType();
3999 MVT EltVT = VT.getVectorElementType();
4000 unsigned NumElts = VT.getVectorNumElements();
4001 unsigned EltSizeInBits = EltVT.getScalarSizeInBits();
4002 SDLoc DL(Op);
4003 SDValue Op0 = Op.getOperand(0);
4004 SDValue Op1 = Op.getOperand(1);
4005 SDValue Op2 = Op.getOperand(2);
4006
4007 if (isa<ConstantSDNode>(Op2))
4008 return Op;
4009
4010 MVT IdxTy = MVT::getIntegerVT(EltSizeInBits);
4011 MVT IdxVTy = MVT::getVectorVT(IdxTy, NumElts);
4012
4013 if (!isTypeLegal(VT) || !isTypeLegal(IdxVTy))
4014 return SDValue();
4015
4016 SDValue SplatElt = DAG.getSplatBuildVector(VT, DL, Op1);
4017 SmallVector<SDValue, 32> RawIndices;
4018 SDValue SplatIdx;
4019 SDValue Indices;
4020
4021 if (!Subtarget.is64Bit() && IdxTy == MVT::i64) {
4022 MVT PairVTy = MVT::getVectorVT(MVT::i32, NumElts * 2);
4023 for (unsigned i = 0; i < NumElts; ++i) {
4024 RawIndices.push_back(Op2);
4025 RawIndices.push_back(DAG.getConstant(0, DL, MVT::i32));
4026 }
4027 SplatIdx = DAG.getBuildVector(PairVTy, DL, RawIndices);
4028 SplatIdx = DAG.getBitcast(IdxVTy, SplatIdx);
4029
4030 RawIndices.clear();
4031 for (unsigned i = 0; i < NumElts; ++i) {
4032 RawIndices.push_back(DAG.getConstant(i, DL, MVT::i32));
4033 RawIndices.push_back(DAG.getConstant(0, DL, MVT::i32));
4034 }
4035 Indices = DAG.getBuildVector(PairVTy, DL, RawIndices);
4036 Indices = DAG.getBitcast(IdxVTy, Indices);
4037 } else {
4038 SplatIdx = DAG.getSplatBuildVector(IdxVTy, DL, Op2);
4039
4040 for (unsigned i = 0; i < NumElts; ++i)
4041 RawIndices.push_back(DAG.getConstant(i, DL, Subtarget.getGRLenVT()));
4042 Indices = DAG.getBuildVector(IdxVTy, DL, RawIndices);
4043 }
4044
4045 // insert vec, elt, idx
4046 // =>
4047 // select (splatidx == {0,1,2...}) ? splatelt : vec
4048 SDValue SelectCC =
4049 DAG.getSetCC(DL, IdxVTy, SplatIdx, Indices, ISD::CondCode::SETEQ);
4050 return DAG.getNode(ISD::VSELECT, DL, VT, SelectCC, SplatElt, Op0);
4051}
4052
4053SDValue LoongArchTargetLowering::lowerATOMIC_FENCE(SDValue Op,
4054 SelectionDAG &DAG) const {
4055 SDLoc DL(Op);
4056 SyncScope::ID FenceSSID =
4057 static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
4058
4059 // singlethread fences only synchronize with signal handlers on the same
4060 // thread and thus only need to preserve instruction order, not actually
4061 // enforce memory ordering.
4062 if (FenceSSID == SyncScope::SingleThread)
4063 // MEMBARRIER is a compiler barrier; it codegens to a no-op.
4064 return DAG.getNode(ISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
4065
4066 return Op;
4067}
4068
4070 MVT GRLenVT, SDValue RMValue) {
4071 // LLVM rounding mode encoding differs from LoongArch FCSR encoding:
4072 // LLVM: 0=RTZ, 1=RNE, 2=RUP, 3=RDN
4073 // FCSR: 0=RNE, 1=RZ, 2=RP, 3=RN
4074 //
4075 // The conversion swaps encodings 0 and 1 while preserving 2 and 3.
4076 // Since the transformation is self-inverse, it applies in both directions:
4077 // LLVM RM <-> LoongArch FCSR RM
4078 //
4079 // Transformation: RM ^ (~(RM >> 1) & 1)
4080 SDValue ShiftRight1 = DAG.getNode(ISD::SRL, DL, GRLenVT, RMValue,
4081 DAG.getConstant(1, DL, GRLenVT));
4082
4083 SDValue SwapMask = DAG.getNode(ISD::AND, DL, GRLenVT,
4084 DAG.getNode(ISD::XOR, DL, GRLenVT, ShiftRight1,
4085 DAG.getConstant(1, DL, GRLenVT)),
4086 DAG.getConstant(1, DL, GRLenVT));
4087
4088 return DAG.getNode(ISD::XOR, DL, GRLenVT, RMValue, SwapMask);
4089}
4090
4091SDValue LoongArchTargetLowering::lowerSET_ROUNDING(SDValue Op,
4092 SelectionDAG &DAG) const {
4093 MVT GRLenVT = Subtarget.getGRLenVT();
4094 SDLoc DL(Op);
4095 SDValue Chain = Op.getOperand(0);
4096 SDValue RMValue = Op.getOperand(1);
4097
4098 if (auto *CVal = dyn_cast<ConstantSDNode>(RMValue)) {
4099 uint64_t RM = CVal->getZExtValue();
4100 if (RM > 3) {
4102 LLVMContext &C = MF.getFunction().getContext();
4103 C.diagnose(DiagnosticInfoUnsupported(
4104 MF.getFunction(),
4105 "rounding mode is not supported by LoongArch hardware",
4106 DiagnosticLocation(DL.getDebugLoc()), DS_Error));
4107 return Chain;
4108 }
4109 }
4110
4111 RMValue = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, RMValue);
4112 RMValue = convertRMEncoding(DAG, DL, GRLenVT, RMValue);
4113
4114 // The RM field in FCSR is at bits [9:8]. Shift the rounding mode value
4115 // into position before writing via WRFCSR.
4116 RMValue = DAG.getNode(ISD::SHL, DL, GRLenVT, RMValue,
4117 DAG.getConstant(8, DL, GRLenVT));
4118
4119 // FCSR3 is an alias of the RM field; writing it avoids clobbering
4120 // unrelated fields in FCSR0.
4121 SDValue FCSRNo = DAG.getTargetConstant(3, DL, GRLenVT);
4122 MachineSDNode *RN = DAG.getMachineNode(LoongArch::WRFCSR, DL, MVT::Other,
4123 FCSRNo, RMValue, Chain);
4124 return SDValue(RN, 0);
4125}
4126
4127SDValue LoongArchTargetLowering::lowerGET_ROUNDING(SDValue Op,
4128 SelectionDAG &DAG) const {
4129 MVT GRLenVT = Subtarget.getGRLenVT();
4130 SDLoc DL(Op);
4131 SDValue Chain = Op->getOperand(0);
4132
4133 // FCSR3 is an alias of the RM field.
4134 SDValue FCSRNo = DAG.getTargetConstant(3, DL, GRLenVT);
4135 MachineSDNode *FCSR = DAG.getMachineNode(LoongArch::RDFCSR, DL, GRLenVT,
4136 MVT::Other, FCSRNo, Chain);
4137 SDValue RMValue = SDValue(FCSR, 0);
4138 Chain = SDValue(FCSR, 1);
4139
4140 // The RM field in FCSR is at bits [9:8].
4141 RMValue = DAG.getNode(ISD::SRL, DL, GRLenVT, RMValue,
4142 DAG.getConstant(8, DL, GRLenVT));
4143 RMValue = convertRMEncoding(DAG, DL, GRLenVT, RMValue);
4144
4145 SDValue RetVal = DAG.getZExtOrTrunc(RMValue, DL, Op.getValueType());
4146 return DAG.getMergeValues({RetVal, Chain}, DL);
4147}
4148
4149SDValue LoongArchTargetLowering::lowerWRITE_REGISTER(SDValue Op,
4150 SelectionDAG &DAG) const {
4151
4152 if (Subtarget.is64Bit() && Op.getOperand(2).getValueType() == MVT::i32) {
4153 DAG.getContext()->emitError(
4154 "On LA64, only 64-bit registers can be written.");
4155 return Op.getOperand(0);
4156 }
4157
4158 if (!Subtarget.is64Bit() && Op.getOperand(2).getValueType() == MVT::i64) {
4159 DAG.getContext()->emitError(
4160 "On LA32, only 32-bit registers can be written.");
4161 return Op.getOperand(0);
4162 }
4163
4164 return Op;
4165}
4166
4167SDValue LoongArchTargetLowering::lowerFRAMEADDR(SDValue Op,
4168 SelectionDAG &DAG) const {
4169 if (!isa<ConstantSDNode>(Op.getOperand(0))) {
4170 DAG.getContext()->emitError("argument to '__builtin_frame_address' must "
4171 "be a constant integer");
4172 return SDValue();
4173 }
4174
4177 Register FrameReg = Subtarget.getRegisterInfo()->getFrameRegister(MF);
4178 EVT VT = Op.getValueType();
4179 SDLoc DL(Op);
4180 SDValue FrameAddr = DAG.getCopyFromReg(DAG.getEntryNode(), DL, FrameReg, VT);
4181 unsigned Depth = Op.getConstantOperandVal(0);
4182 int GRLenInBytes = Subtarget.getGRLen() / 8;
4183
4184 while (Depth--) {
4185 int Offset = -(GRLenInBytes * 2);
4186 SDValue Ptr = DAG.getNode(ISD::ADD, DL, VT, FrameAddr,
4187 DAG.getSignedConstant(Offset, DL, VT));
4188 FrameAddr =
4189 DAG.getLoad(VT, DL, DAG.getEntryNode(), Ptr, MachinePointerInfo());
4190 }
4191 return FrameAddr;
4192}
4193
4194SDValue LoongArchTargetLowering::lowerRETURNADDR(SDValue Op,
4195 SelectionDAG &DAG) const {
4196 // Currently only support lowering return address for current frame.
4197 if (Op.getConstantOperandVal(0) != 0) {
4198 DAG.getContext()->emitError(
4199 "return address can only be determined for the current frame");
4200 return SDValue();
4201 }
4202
4205 MVT GRLenVT = Subtarget.getGRLenVT();
4206
4207 // Return the value of the return address register, marking it an implicit
4208 // live-in.
4209 Register Reg = MF.addLiveIn(Subtarget.getRegisterInfo()->getRARegister(),
4210 getRegClassFor(GRLenVT));
4211 return DAG.getCopyFromReg(DAG.getEntryNode(), SDLoc(Op), Reg, GRLenVT);
4212}
4213
4214SDValue LoongArchTargetLowering::lowerEH_DWARF_CFA(SDValue Op,
4215 SelectionDAG &DAG) const {
4217 auto Size = Subtarget.getGRLen() / 8;
4218 auto FI = MF.getFrameInfo().CreateFixedObject(Size, 0, false);
4219 return DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
4220}
4221
4222SDValue LoongArchTargetLowering::lowerVASTART(SDValue Op,
4223 SelectionDAG &DAG) const {
4225 auto *FuncInfo = MF.getInfo<LoongArchMachineFunctionInfo>();
4226
4227 SDLoc DL(Op);
4228 SDValue FI = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(),
4230
4231 // vastart just stores the address of the VarArgsFrameIndex slot into the
4232 // memory location argument.
4233 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4234 return DAG.getStore(Op.getOperand(0), DL, FI, Op.getOperand(1),
4235 MachinePointerInfo(SV));
4236}
4237
4238SDValue LoongArchTargetLowering::lowerUINT_TO_FP(SDValue Op,
4239 SelectionDAG &DAG) const {
4240 SDLoc DL(Op);
4241 SDValue Op0 = Op.getOperand(0);
4242 EVT VT = Op.getValueType();
4243 EVT Op0VT = Op0.getValueType();
4244
4245 if (VT.isVector()) {
4246 if (VT.getScalarSizeInBits() != Op0VT.getScalarSizeInBits())
4247 return SDValue();
4248 return Op;
4249 }
4250
4251 if ((DAG.SignBitIsZero(Op0) || Op->getFlags().hasNonNeg()) &&
4254 return DAG.getNode(ISD::SINT_TO_FP, DL, VT, Op0);
4255
4256 // We can't do uint64 -> double -> float because of double-rounding issue.
4257 if (Subtarget.hasExtLSX() && Op0VT == MVT::i64 && VT == MVT::f64) {
4258 Op0 = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2i64, Op0);
4259 SDValue Conv = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::v2f64, Op0);
4260 Conv = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f64, Conv,
4261 DAG.getIntPtrConstant(0, DL));
4262 return Conv;
4263 }
4264
4265 if (!Subtarget.is64Bit() || !Subtarget.hasBasicF() || Subtarget.hasBasicD())
4266 return SDValue();
4267
4268 assert(Subtarget.is64Bit() && Subtarget.hasBasicF() &&
4269 !Subtarget.hasBasicD() && "unexpected target features");
4270
4271 if (Op0->getOpcode() == ISD::AND) {
4272 auto *C = dyn_cast<ConstantSDNode>(Op0.getOperand(1));
4273 if (C && C->getZExtValue() < UINT64_C(0xFFFFFFFF))
4274 return Op;
4275 }
4276
4277 if (Op0->getOpcode() == LoongArchISD::BSTRPICK &&
4278 Op0.getConstantOperandVal(1) < UINT64_C(0X1F) &&
4279 Op0.getConstantOperandVal(2) == UINT64_C(0))
4280 return Op;
4281
4282 if (Op0.getOpcode() == ISD::AssertZext &&
4283 dyn_cast<VTSDNode>(Op0.getOperand(1))->getVT().bitsLT(MVT::i32))
4284 return Op;
4285
4286 EVT OpVT = Op0.getValueType();
4287 EVT RetVT = Op.getValueType();
4288 RTLIB::Libcall LC = RTLIB::getUINTTOFP(OpVT, RetVT);
4289 MakeLibCallOptions CallOptions;
4290 CallOptions.setTypeListBeforeSoften(OpVT, RetVT);
4291 SDValue Chain = SDValue();
4292 SDValue Result;
4293 std::tie(Result, Chain) =
4294 makeLibCall(DAG, LC, Op.getValueType(), Op0, CallOptions, DL, Chain);
4295 return Result;
4296}
4297
4298SDValue LoongArchTargetLowering::lowerSINT_TO_FP(SDValue Op,
4299 SelectionDAG &DAG) const {
4300 assert(Subtarget.is64Bit() && Subtarget.hasBasicF() &&
4301 !Subtarget.hasBasicD() && "unexpected target features");
4302
4303 SDLoc DL(Op);
4304 SDValue Op0 = Op.getOperand(0);
4305
4306 if ((Op0.getOpcode() == ISD::AssertSext ||
4308 dyn_cast<VTSDNode>(Op0.getOperand(1))->getVT().bitsLE(MVT::i32))
4309 return Op;
4310
4311 EVT OpVT = Op0.getValueType();
4312 EVT RetVT = Op.getValueType();
4313 RTLIB::Libcall LC = RTLIB::getSINTTOFP(OpVT, RetVT);
4314 MakeLibCallOptions CallOptions;
4315 CallOptions.setTypeListBeforeSoften(OpVT, RetVT);
4316 SDValue Chain = SDValue();
4317 SDValue Result;
4318 std::tie(Result, Chain) =
4319 makeLibCall(DAG, LC, Op.getValueType(), Op0, CallOptions, DL, Chain);
4320 return Result;
4321}
4322
4323SDValue LoongArchTargetLowering::lowerBITCAST(SDValue Op,
4324 SelectionDAG &DAG) const {
4325
4326 SDLoc DL(Op);
4327 EVT VT = Op.getValueType();
4328 SDValue Op0 = Op.getOperand(0);
4329 EVT Op0VT = Op0.getValueType();
4330
4331 if (Op.getValueType() == MVT::f32 && Op0VT == MVT::i32 &&
4332 Subtarget.is64Bit() && Subtarget.hasBasicF()) {
4333 SDValue NewOp0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
4334 return DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64, DL, MVT::f32, NewOp0);
4335 }
4336 if (VT == MVT::f64 && Op0VT == MVT::i64 && !Subtarget.is64Bit()) {
4337 SDValue Lo, Hi;
4338 std::tie(Lo, Hi) = DAG.SplitScalar(Op0, DL, MVT::i32, MVT::i32);
4339 return DAG.getNode(LoongArchISD::BUILD_PAIR_F64, DL, MVT::f64, Lo, Hi);
4340 }
4341 return Op;
4342}
4343
4344SDValue LoongArchTargetLowering::lowerFP_TO_SINT(SDValue Op,
4345 SelectionDAG &DAG) const {
4346
4347 SDLoc DL(Op);
4348 SDValue Op0 = Op.getOperand(0);
4349
4350 if (Op0.getValueType() == MVT::f16)
4351 Op0 = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Op0);
4352
4353 if (Op.getValueSizeInBits() > 32 && Subtarget.hasBasicF() &&
4354 !Subtarget.hasBasicD()) {
4355 SDValue Dst = DAG.getNode(LoongArchISD::FTINT, DL, MVT::f32, Op0);
4356 return DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Dst);
4357 }
4358
4359 EVT FPTy = EVT::getFloatingPointVT(Op.getValueSizeInBits());
4360 SDValue Trunc = DAG.getNode(LoongArchISD::FTINT, DL, FPTy, Op0);
4361 return DAG.getNode(ISD::BITCAST, DL, Op.getValueType(), Trunc);
4362}
4363
4364SDValue LoongArchTargetLowering::lowerFP_TO_UINT(SDValue Op,
4365 SelectionDAG &DAG) const {
4366 if (!Subtarget.hasExtLSX())
4367 return SDValue();
4368
4369 SDLoc DL(Op);
4370 SDValue Src = Op.getOperand(0);
4371 EVT VT = Op.getValueType();
4372 EVT SrcVT = Src.getValueType();
4373
4374 if (VT != MVT::i64)
4375 return SDValue();
4376
4377 if (SrcVT != MVT::f32 && SrcVT != MVT::f64)
4378 return SDValue();
4379
4380 if (SrcVT == MVT::f32)
4381 Src = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f64, Src);
4382 Src = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2f64, Src);
4383 SDValue Conv = DAG.getNode(ISD::FP_TO_UINT, DL, MVT::v2i64, Src);
4384 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VT, Conv,
4385 DAG.getIntPtrConstant(0, DL));
4386}
4387
4389 SelectionDAG &DAG, unsigned Flags) {
4390 return DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, Flags);
4391}
4392
4394 SelectionDAG &DAG, unsigned Flags) {
4395 return DAG.getTargetBlockAddress(N->getBlockAddress(), Ty, N->getOffset(),
4396 Flags);
4397}
4398
4400 SelectionDAG &DAG, unsigned Flags) {
4401 return DAG.getTargetConstantPool(N->getConstVal(), Ty, N->getAlign(),
4402 N->getOffset(), Flags);
4403}
4404
4406 SelectionDAG &DAG, unsigned Flags) {
4407 return DAG.getTargetJumpTable(N->getIndex(), Ty, Flags);
4408}
4409
4410template <class NodeTy>
4411SDValue LoongArchTargetLowering::getAddr(NodeTy *N, SelectionDAG &DAG,
4413 bool IsLocal) const {
4414 SDLoc DL(N);
4415 EVT Ty = getPointerTy(DAG.getDataLayout());
4416 SDValue Addr = getTargetNode(N, DL, Ty, DAG, 0);
4417 SDValue Load;
4418
4419 switch (M) {
4420 default:
4421 report_fatal_error("Unsupported code model");
4422
4423 case CodeModel::Large: {
4424 assert(Subtarget.is64Bit() && "Large code model requires LA64");
4425
4426 // This is not actually used, but is necessary for successfully matching
4427 // the PseudoLA_*_LARGE nodes.
4428 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4429 if (IsLocal) {
4430 // This generates the pattern (PseudoLA_PCREL_LARGE tmp sym), that
4431 // eventually becomes the desired 5-insn code sequence.
4432 Load = SDValue(DAG.getMachineNode(LoongArch::PseudoLA_PCREL_LARGE, DL, Ty,
4433 Tmp, Addr),
4434 0);
4435 } else {
4436 // This generates the pattern (PseudoLA_GOT_LARGE tmp sym), that
4437 // eventually becomes the desired 5-insn code sequence.
4438 Load = SDValue(
4439 DAG.getMachineNode(LoongArch::PseudoLA_GOT_LARGE, DL, Ty, Tmp, Addr),
4440 0);
4441 }
4442 break;
4443 }
4444
4445 case CodeModel::Small:
4446 case CodeModel::Medium:
4447 if (IsLocal) {
4448 // This generates the pattern (PseudoLA_PCREL sym), which
4449 //
4450 // for la32r expands to:
4451 // (addi.w (pcaddu12i %pcadd_hi20(sym)) %pcadd_lo12(.Lpcadd_hi)).
4452 //
4453 // for la32s and la64 expands to:
4454 // (addi.w/d (pcalau12i %pc_hi20(sym)) %pc_lo12(sym)).
4455 Load = SDValue(
4456 DAG.getMachineNode(LoongArch::PseudoLA_PCREL, DL, Ty, Addr), 0);
4457 } else {
4458 // This generates the pattern (PseudoLA_GOT sym), which
4459 //
4460 // for la32r expands to:
4461 // (ld.w (pcaddu12i %got_pcadd_hi20(sym)) %pcadd_lo12(.Lpcadd_hi)).
4462 //
4463 // for la32s and la64 expands to:
4464 // (ld.w/d (pcalau12i %got_pc_hi20(sym)) %got_pc_lo12(sym)).
4465 Load =
4466 SDValue(DAG.getMachineNode(LoongArch::PseudoLA_GOT, DL, Ty, Addr), 0);
4467 }
4468 }
4469
4470 if (!IsLocal) {
4471 // Mark the load instruction as invariant to enable hoisting in MachineLICM.
4473 MachineMemOperand *MemOp = MF.getMachineMemOperand(
4477 LLT(Ty.getSimpleVT()), Align(Ty.getFixedSizeInBits() / 8));
4478 DAG.setNodeMemRefs(cast<MachineSDNode>(Load.getNode()), {MemOp});
4479 }
4480
4481 return Load;
4482}
4483
4484SDValue LoongArchTargetLowering::lowerBlockAddress(SDValue Op,
4485 SelectionDAG &DAG) const {
4486 return getAddr(cast<BlockAddressSDNode>(Op), DAG,
4487 DAG.getTarget().getCodeModel());
4488}
4489
4490SDValue LoongArchTargetLowering::lowerJumpTable(SDValue Op,
4491 SelectionDAG &DAG) const {
4492 return getAddr(cast<JumpTableSDNode>(Op), DAG,
4493 DAG.getTarget().getCodeModel());
4494}
4495
4496SDValue LoongArchTargetLowering::lowerConstantPool(SDValue Op,
4497 SelectionDAG &DAG) const {
4498 return getAddr(cast<ConstantPoolSDNode>(Op), DAG,
4499 DAG.getTarget().getCodeModel());
4500}
4501
4502SDValue LoongArchTargetLowering::lowerGlobalAddress(SDValue Op,
4503 SelectionDAG &DAG) const {
4504 GlobalAddressSDNode *N = cast<GlobalAddressSDNode>(Op);
4505 assert(N->getOffset() == 0 && "unexpected offset in global node");
4506 auto CM = DAG.getTarget().getCodeModel();
4507 const GlobalValue *GV = N->getGlobal();
4508
4509 if (GV->isDSOLocal() && isa<GlobalVariable>(GV)) {
4510 if (auto GCM = dyn_cast<GlobalVariable>(GV)->getCodeModel())
4511 CM = *GCM;
4512 }
4513
4514 return getAddr(N, DAG, CM, GV->isDSOLocal());
4515}
4516
4517SDValue LoongArchTargetLowering::getStaticTLSAddr(GlobalAddressSDNode *N,
4518 SelectionDAG &DAG,
4519 unsigned Opc, bool UseGOT,
4520 bool Large) const {
4521 SDLoc DL(N);
4522 EVT Ty = getPointerTy(DAG.getDataLayout());
4523 MVT GRLenVT = Subtarget.getGRLenVT();
4524
4525 // This is not actually used, but is necessary for successfully matching the
4526 // PseudoLA_*_LARGE nodes.
4527 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4528 SDValue Addr = DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, 0);
4529
4530 // Only IE needs an extra argument for large code model.
4531 SDValue Offset = Opc == LoongArch::PseudoLA_TLS_IE_LARGE
4532 ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4533 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4534
4535 // If it is LE for normal/medium code model, the add tp operation will occur
4536 // during the pseudo-instruction expansion.
4537 if (Opc == LoongArch::PseudoLA_TLS_LE && !Large)
4538 return Offset;
4539
4540 if (UseGOT) {
4541 // Mark the load instruction as invariant to enable hoisting in MachineLICM.
4543 MachineMemOperand *MemOp = MF.getMachineMemOperand(
4547 LLT(Ty.getSimpleVT()), Align(Ty.getFixedSizeInBits() / 8));
4548 DAG.setNodeMemRefs(cast<MachineSDNode>(Offset.getNode()), {MemOp});
4549 }
4550
4551 // Add the thread pointer.
4552 return DAG.getNode(ISD::ADD, DL, Ty, Offset,
4553 DAG.getRegister(LoongArch::R2, GRLenVT));
4554}
4555
4556SDValue LoongArchTargetLowering::getDynamicTLSAddr(GlobalAddressSDNode *N,
4557 SelectionDAG &DAG,
4558 unsigned Opc,
4559 bool Large) const {
4560 SDLoc DL(N);
4561 EVT Ty = getPointerTy(DAG.getDataLayout());
4562 IntegerType *CallTy = Type::getIntNTy(*DAG.getContext(), Ty.getSizeInBits());
4563
4564 // This is not actually used, but is necessary for successfully matching the
4565 // PseudoLA_*_LARGE nodes.
4566 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4567
4568 // Use a PC-relative addressing mode to access the dynamic GOT address.
4569 SDValue Addr = DAG.getTargetGlobalAddress(N->getGlobal(), DL, Ty, 0, 0);
4570 SDValue Load = Large ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4571 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4572
4573 // Prepare argument list to generate call.
4575 Args.emplace_back(Load, CallTy);
4576
4577 // Setup call to __tls_get_addr.
4578 TargetLowering::CallLoweringInfo CLI(DAG);
4579 CLI.setDebugLoc(DL)
4580 .setChain(DAG.getEntryNode())
4581 .setLibCallee(CallingConv::C, CallTy,
4582 DAG.getExternalSymbol("__tls_get_addr", Ty),
4583 std::move(Args));
4584
4585 return LowerCallTo(CLI).first;
4586}
4587
4588SDValue LoongArchTargetLowering::getTLSDescAddr(GlobalAddressSDNode *N,
4589 SelectionDAG &DAG, unsigned Opc,
4590 bool Large) const {
4591 SDLoc DL(N);
4592 EVT Ty = getPointerTy(DAG.getDataLayout());
4593 const GlobalValue *GV = N->getGlobal();
4594
4595 // This is not actually used, but is necessary for successfully matching the
4596 // PseudoLA_*_LARGE nodes.
4597 SDValue Tmp = DAG.getConstant(0, DL, Ty);
4598
4599 // Use a PC-relative addressing mode to access the global dynamic GOT address.
4600 // This generates the pattern (PseudoLA_TLS_DESC_PC{,LARGE} sym).
4601 SDValue Addr = DAG.getTargetGlobalAddress(GV, DL, Ty, 0, 0);
4602 return Large ? SDValue(DAG.getMachineNode(Opc, DL, Ty, Tmp, Addr), 0)
4603 : SDValue(DAG.getMachineNode(Opc, DL, Ty, Addr), 0);
4604}
4605
4606SDValue
4607LoongArchTargetLowering::lowerGlobalTLSAddress(SDValue Op,
4608 SelectionDAG &DAG) const {
4611 report_fatal_error("In GHC calling convention TLS is not supported");
4612
4613 bool Large = DAG.getTarget().getCodeModel() == CodeModel::Large;
4614 assert((!Large || Subtarget.is64Bit()) && "Large code model requires LA64");
4615
4616 GlobalAddressSDNode *N = cast<GlobalAddressSDNode>(Op);
4617 assert(N->getOffset() == 0 && "unexpected offset in global node");
4618
4619 if (DAG.getTarget().useEmulatedTLS())
4620 reportFatalUsageError("the emulated TLS is prohibited");
4621
4622 bool IsDesc = DAG.getTarget().useTLSDESC();
4623
4624 switch (getTargetMachine().getTLSModel(N->getGlobal())) {
4626 // In this model, application code calls the dynamic linker function
4627 // __tls_get_addr to locate TLS offsets into the dynamic thread vector at
4628 // runtime.
4629 if (!IsDesc)
4630 return getDynamicTLSAddr(N, DAG,
4631 Large ? LoongArch::PseudoLA_TLS_GD_LARGE
4632 : LoongArch::PseudoLA_TLS_GD,
4633 Large);
4634 break;
4636 // Same as GeneralDynamic, except for assembly modifiers and relocation
4637 // records.
4638 if (!IsDesc)
4639 return getDynamicTLSAddr(N, DAG,
4640 Large ? LoongArch::PseudoLA_TLS_LD_LARGE
4641 : LoongArch::PseudoLA_TLS_LD,
4642 Large);
4643 break;
4645 // This model uses the GOT to resolve TLS offsets.
4646 return getStaticTLSAddr(N, DAG,
4647 Large ? LoongArch::PseudoLA_TLS_IE_LARGE
4648 : LoongArch::PseudoLA_TLS_IE,
4649 /*UseGOT=*/true, Large);
4651 // This model is used when static linking as the TLS offsets are resolved
4652 // during program linking.
4653 //
4654 // This node doesn't need an extra argument for the large code model.
4655 return getStaticTLSAddr(N, DAG, LoongArch::PseudoLA_TLS_LE,
4656 /*UseGOT=*/false, Large);
4657 }
4658
4659 return getTLSDescAddr(N, DAG,
4660 Large ? LoongArch::PseudoLA_TLS_DESC_LARGE
4661 : LoongArch::PseudoLA_TLS_DESC,
4662 Large);
4663}
4664
4665template <unsigned N>
4667 SelectionDAG &DAG, bool IsSigned = false) {
4668 auto *CImm = cast<ConstantSDNode>(Op->getOperand(ImmOp));
4669 // Check the ImmArg.
4670 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
4671 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
4672 DAG.getContext()->emitError(Op->getOperationName(0) +
4673 ": argument out of range.");
4674 return DAG.getNode(ISD::UNDEF, SDLoc(Op), Op.getValueType());
4675 }
4676 return SDValue();
4677}
4678
4679SDValue
4680LoongArchTargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
4681 SelectionDAG &DAG) const {
4682 switch (Op.getConstantOperandVal(0)) {
4683 default:
4684 return SDValue(); // Don't custom lower most intrinsics.
4685 case Intrinsic::thread_pointer: {
4686 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4687 return DAG.getRegister(LoongArch::R2, PtrVT);
4688 }
4689 case Intrinsic::loongarch_lsx_vpickve2gr_d:
4690 case Intrinsic::loongarch_lsx_vpickve2gr_du:
4691 case Intrinsic::loongarch_lsx_vreplvei_d:
4692 case Intrinsic::loongarch_lasx_xvrepl128vei_d:
4693 return checkIntrinsicImmArg<1>(Op, 2, DAG);
4694 case Intrinsic::loongarch_lsx_vreplvei_w:
4695 case Intrinsic::loongarch_lasx_xvrepl128vei_w:
4696 case Intrinsic::loongarch_lasx_xvpickve2gr_d:
4697 case Intrinsic::loongarch_lasx_xvpickve2gr_du:
4698 case Intrinsic::loongarch_lasx_xvpickve_d:
4699 case Intrinsic::loongarch_lasx_xvpickve_d_f:
4700 return checkIntrinsicImmArg<2>(Op, 2, DAG);
4701 case Intrinsic::loongarch_lasx_xvinsve0_d:
4702 return checkIntrinsicImmArg<2>(Op, 3, DAG);
4703 case Intrinsic::loongarch_lsx_vsat_b:
4704 case Intrinsic::loongarch_lsx_vsat_bu:
4705 case Intrinsic::loongarch_lsx_vrotri_b:
4706 case Intrinsic::loongarch_lsx_vsllwil_h_b:
4707 case Intrinsic::loongarch_lsx_vsllwil_hu_bu:
4708 case Intrinsic::loongarch_lsx_vsrlri_b:
4709 case Intrinsic::loongarch_lsx_vsrari_b:
4710 case Intrinsic::loongarch_lsx_vreplvei_h:
4711 case Intrinsic::loongarch_lasx_xvsat_b:
4712 case Intrinsic::loongarch_lasx_xvsat_bu:
4713 case Intrinsic::loongarch_lasx_xvrotri_b:
4714 case Intrinsic::loongarch_lasx_xvsllwil_h_b:
4715 case Intrinsic::loongarch_lasx_xvsllwil_hu_bu:
4716 case Intrinsic::loongarch_lasx_xvsrlri_b:
4717 case Intrinsic::loongarch_lasx_xvsrari_b:
4718 case Intrinsic::loongarch_lasx_xvrepl128vei_h:
4719 case Intrinsic::loongarch_lasx_xvpickve_w:
4720 case Intrinsic::loongarch_lasx_xvpickve_w_f:
4721 return checkIntrinsicImmArg<3>(Op, 2, DAG);
4722 case Intrinsic::loongarch_lasx_xvinsve0_w:
4723 return checkIntrinsicImmArg<3>(Op, 3, DAG);
4724 case Intrinsic::loongarch_lsx_vsat_h:
4725 case Intrinsic::loongarch_lsx_vsat_hu:
4726 case Intrinsic::loongarch_lsx_vrotri_h:
4727 case Intrinsic::loongarch_lsx_vsllwil_w_h:
4728 case Intrinsic::loongarch_lsx_vsllwil_wu_hu:
4729 case Intrinsic::loongarch_lsx_vsrlri_h:
4730 case Intrinsic::loongarch_lsx_vsrari_h:
4731 case Intrinsic::loongarch_lsx_vreplvei_b:
4732 case Intrinsic::loongarch_lasx_xvsat_h:
4733 case Intrinsic::loongarch_lasx_xvsat_hu:
4734 case Intrinsic::loongarch_lasx_xvrotri_h:
4735 case Intrinsic::loongarch_lasx_xvsllwil_w_h:
4736 case Intrinsic::loongarch_lasx_xvsllwil_wu_hu:
4737 case Intrinsic::loongarch_lasx_xvsrlri_h:
4738 case Intrinsic::loongarch_lasx_xvsrari_h:
4739 case Intrinsic::loongarch_lasx_xvrepl128vei_b:
4740 return checkIntrinsicImmArg<4>(Op, 2, DAG);
4741 case Intrinsic::loongarch_lsx_vsrlni_b_h:
4742 case Intrinsic::loongarch_lsx_vsrani_b_h:
4743 case Intrinsic::loongarch_lsx_vsrlrni_b_h:
4744 case Intrinsic::loongarch_lsx_vsrarni_b_h:
4745 case Intrinsic::loongarch_lsx_vssrlni_b_h:
4746 case Intrinsic::loongarch_lsx_vssrani_b_h:
4747 case Intrinsic::loongarch_lsx_vssrlni_bu_h:
4748 case Intrinsic::loongarch_lsx_vssrani_bu_h:
4749 case Intrinsic::loongarch_lsx_vssrlrni_b_h:
4750 case Intrinsic::loongarch_lsx_vssrarni_b_h:
4751 case Intrinsic::loongarch_lsx_vssrlrni_bu_h:
4752 case Intrinsic::loongarch_lsx_vssrarni_bu_h:
4753 case Intrinsic::loongarch_lasx_xvsrlni_b_h:
4754 case Intrinsic::loongarch_lasx_xvsrani_b_h:
4755 case Intrinsic::loongarch_lasx_xvsrlrni_b_h:
4756 case Intrinsic::loongarch_lasx_xvsrarni_b_h:
4757 case Intrinsic::loongarch_lasx_xvssrlni_b_h:
4758 case Intrinsic::loongarch_lasx_xvssrani_b_h:
4759 case Intrinsic::loongarch_lasx_xvssrlni_bu_h:
4760 case Intrinsic::loongarch_lasx_xvssrani_bu_h:
4761 case Intrinsic::loongarch_lasx_xvssrlrni_b_h:
4762 case Intrinsic::loongarch_lasx_xvssrarni_b_h:
4763 case Intrinsic::loongarch_lasx_xvssrlrni_bu_h:
4764 case Intrinsic::loongarch_lasx_xvssrarni_bu_h:
4765 return checkIntrinsicImmArg<4>(Op, 3, DAG);
4766 case Intrinsic::loongarch_lsx_vsat_w:
4767 case Intrinsic::loongarch_lsx_vsat_wu:
4768 case Intrinsic::loongarch_lsx_vrotri_w:
4769 case Intrinsic::loongarch_lsx_vsllwil_d_w:
4770 case Intrinsic::loongarch_lsx_vsllwil_du_wu:
4771 case Intrinsic::loongarch_lsx_vsrlri_w:
4772 case Intrinsic::loongarch_lsx_vsrari_w:
4773 case Intrinsic::loongarch_lsx_vslei_bu:
4774 case Intrinsic::loongarch_lsx_vslei_hu:
4775 case Intrinsic::loongarch_lsx_vslei_wu:
4776 case Intrinsic::loongarch_lsx_vslei_du:
4777 case Intrinsic::loongarch_lsx_vslti_bu:
4778 case Intrinsic::loongarch_lsx_vslti_hu:
4779 case Intrinsic::loongarch_lsx_vslti_wu:
4780 case Intrinsic::loongarch_lsx_vslti_du:
4781 case Intrinsic::loongarch_lsx_vbsll_v:
4782 case Intrinsic::loongarch_lsx_vbsrl_v:
4783 case Intrinsic::loongarch_lasx_xvsat_w:
4784 case Intrinsic::loongarch_lasx_xvsat_wu:
4785 case Intrinsic::loongarch_lasx_xvrotri_w:
4786 case Intrinsic::loongarch_lasx_xvsllwil_d_w:
4787 case Intrinsic::loongarch_lasx_xvsllwil_du_wu:
4788 case Intrinsic::loongarch_lasx_xvsrlri_w:
4789 case Intrinsic::loongarch_lasx_xvsrari_w:
4790 case Intrinsic::loongarch_lasx_xvslei_bu:
4791 case Intrinsic::loongarch_lasx_xvslei_hu:
4792 case Intrinsic::loongarch_lasx_xvslei_wu:
4793 case Intrinsic::loongarch_lasx_xvslei_du:
4794 case Intrinsic::loongarch_lasx_xvslti_bu:
4795 case Intrinsic::loongarch_lasx_xvslti_hu:
4796 case Intrinsic::loongarch_lasx_xvslti_wu:
4797 case Intrinsic::loongarch_lasx_xvslti_du:
4798 case Intrinsic::loongarch_lasx_xvbsll_v:
4799 case Intrinsic::loongarch_lasx_xvbsrl_v:
4800 return checkIntrinsicImmArg<5>(Op, 2, DAG);
4801 case Intrinsic::loongarch_lsx_vseqi_b:
4802 case Intrinsic::loongarch_lsx_vseqi_h:
4803 case Intrinsic::loongarch_lsx_vseqi_w:
4804 case Intrinsic::loongarch_lsx_vseqi_d:
4805 case Intrinsic::loongarch_lsx_vslei_b:
4806 case Intrinsic::loongarch_lsx_vslei_h:
4807 case Intrinsic::loongarch_lsx_vslei_w:
4808 case Intrinsic::loongarch_lsx_vslei_d:
4809 case Intrinsic::loongarch_lsx_vslti_b:
4810 case Intrinsic::loongarch_lsx_vslti_h:
4811 case Intrinsic::loongarch_lsx_vslti_w:
4812 case Intrinsic::loongarch_lsx_vslti_d:
4813 case Intrinsic::loongarch_lasx_xvseqi_b:
4814 case Intrinsic::loongarch_lasx_xvseqi_h:
4815 case Intrinsic::loongarch_lasx_xvseqi_w:
4816 case Intrinsic::loongarch_lasx_xvseqi_d:
4817 case Intrinsic::loongarch_lasx_xvslei_b:
4818 case Intrinsic::loongarch_lasx_xvslei_h:
4819 case Intrinsic::loongarch_lasx_xvslei_w:
4820 case Intrinsic::loongarch_lasx_xvslei_d:
4821 case Intrinsic::loongarch_lasx_xvslti_b:
4822 case Intrinsic::loongarch_lasx_xvslti_h:
4823 case Intrinsic::loongarch_lasx_xvslti_w:
4824 case Intrinsic::loongarch_lasx_xvslti_d:
4825 return checkIntrinsicImmArg<5>(Op, 2, DAG, /*IsSigned=*/true);
4826 case Intrinsic::loongarch_lsx_vsrlni_h_w:
4827 case Intrinsic::loongarch_lsx_vsrani_h_w:
4828 case Intrinsic::loongarch_lsx_vsrlrni_h_w:
4829 case Intrinsic::loongarch_lsx_vsrarni_h_w:
4830 case Intrinsic::loongarch_lsx_vssrlni_h_w:
4831 case Intrinsic::loongarch_lsx_vssrani_h_w:
4832 case Intrinsic::loongarch_lsx_vssrlni_hu_w:
4833 case Intrinsic::loongarch_lsx_vssrani_hu_w:
4834 case Intrinsic::loongarch_lsx_vssrlrni_h_w:
4835 case Intrinsic::loongarch_lsx_vssrarni_h_w:
4836 case Intrinsic::loongarch_lsx_vssrlrni_hu_w:
4837 case Intrinsic::loongarch_lsx_vssrarni_hu_w:
4838 case Intrinsic::loongarch_lsx_vfrstpi_b:
4839 case Intrinsic::loongarch_lsx_vfrstpi_h:
4840 case Intrinsic::loongarch_lasx_xvsrlni_h_w:
4841 case Intrinsic::loongarch_lasx_xvsrani_h_w:
4842 case Intrinsic::loongarch_lasx_xvsrlrni_h_w:
4843 case Intrinsic::loongarch_lasx_xvsrarni_h_w:
4844 case Intrinsic::loongarch_lasx_xvssrlni_h_w:
4845 case Intrinsic::loongarch_lasx_xvssrani_h_w:
4846 case Intrinsic::loongarch_lasx_xvssrlni_hu_w:
4847 case Intrinsic::loongarch_lasx_xvssrani_hu_w:
4848 case Intrinsic::loongarch_lasx_xvssrlrni_h_w:
4849 case Intrinsic::loongarch_lasx_xvssrarni_h_w:
4850 case Intrinsic::loongarch_lasx_xvssrlrni_hu_w:
4851 case Intrinsic::loongarch_lasx_xvssrarni_hu_w:
4852 case Intrinsic::loongarch_lasx_xvfrstpi_b:
4853 case Intrinsic::loongarch_lasx_xvfrstpi_h:
4854 return checkIntrinsicImmArg<5>(Op, 3, DAG);
4855 case Intrinsic::loongarch_lsx_vsat_d:
4856 case Intrinsic::loongarch_lsx_vsat_du:
4857 case Intrinsic::loongarch_lsx_vrotri_d:
4858 case Intrinsic::loongarch_lsx_vsrlri_d:
4859 case Intrinsic::loongarch_lsx_vsrari_d:
4860 case Intrinsic::loongarch_lasx_xvsat_d:
4861 case Intrinsic::loongarch_lasx_xvsat_du:
4862 case Intrinsic::loongarch_lasx_xvrotri_d:
4863 case Intrinsic::loongarch_lasx_xvsrlri_d:
4864 case Intrinsic::loongarch_lasx_xvsrari_d:
4865 return checkIntrinsicImmArg<6>(Op, 2, DAG);
4866 case Intrinsic::loongarch_lsx_vsrlni_w_d:
4867 case Intrinsic::loongarch_lsx_vsrani_w_d:
4868 case Intrinsic::loongarch_lsx_vsrlrni_w_d:
4869 case Intrinsic::loongarch_lsx_vsrarni_w_d:
4870 case Intrinsic::loongarch_lsx_vssrlni_w_d:
4871 case Intrinsic::loongarch_lsx_vssrani_w_d:
4872 case Intrinsic::loongarch_lsx_vssrlni_wu_d:
4873 case Intrinsic::loongarch_lsx_vssrani_wu_d:
4874 case Intrinsic::loongarch_lsx_vssrlrni_w_d:
4875 case Intrinsic::loongarch_lsx_vssrarni_w_d:
4876 case Intrinsic::loongarch_lsx_vssrlrni_wu_d:
4877 case Intrinsic::loongarch_lsx_vssrarni_wu_d:
4878 case Intrinsic::loongarch_lasx_xvsrlni_w_d:
4879 case Intrinsic::loongarch_lasx_xvsrani_w_d:
4880 case Intrinsic::loongarch_lasx_xvsrlrni_w_d:
4881 case Intrinsic::loongarch_lasx_xvsrarni_w_d:
4882 case Intrinsic::loongarch_lasx_xvssrlni_w_d:
4883 case Intrinsic::loongarch_lasx_xvssrani_w_d:
4884 case Intrinsic::loongarch_lasx_xvssrlni_wu_d:
4885 case Intrinsic::loongarch_lasx_xvssrani_wu_d:
4886 case Intrinsic::loongarch_lasx_xvssrlrni_w_d:
4887 case Intrinsic::loongarch_lasx_xvssrarni_w_d:
4888 case Intrinsic::loongarch_lasx_xvssrlrni_wu_d:
4889 case Intrinsic::loongarch_lasx_xvssrarni_wu_d:
4890 return checkIntrinsicImmArg<6>(Op, 3, DAG);
4891 case Intrinsic::loongarch_lsx_vsrlni_d_q:
4892 case Intrinsic::loongarch_lsx_vsrani_d_q:
4893 case Intrinsic::loongarch_lsx_vsrlrni_d_q:
4894 case Intrinsic::loongarch_lsx_vsrarni_d_q:
4895 case Intrinsic::loongarch_lsx_vssrlni_d_q:
4896 case Intrinsic::loongarch_lsx_vssrani_d_q:
4897 case Intrinsic::loongarch_lsx_vssrlni_du_q:
4898 case Intrinsic::loongarch_lsx_vssrani_du_q:
4899 case Intrinsic::loongarch_lsx_vssrlrni_d_q:
4900 case Intrinsic::loongarch_lsx_vssrarni_d_q:
4901 case Intrinsic::loongarch_lsx_vssrlrni_du_q:
4902 case Intrinsic::loongarch_lsx_vssrarni_du_q:
4903 case Intrinsic::loongarch_lasx_xvsrlni_d_q:
4904 case Intrinsic::loongarch_lasx_xvsrani_d_q:
4905 case Intrinsic::loongarch_lasx_xvsrlrni_d_q:
4906 case Intrinsic::loongarch_lasx_xvsrarni_d_q:
4907 case Intrinsic::loongarch_lasx_xvssrlni_d_q:
4908 case Intrinsic::loongarch_lasx_xvssrani_d_q:
4909 case Intrinsic::loongarch_lasx_xvssrlni_du_q:
4910 case Intrinsic::loongarch_lasx_xvssrani_du_q:
4911 case Intrinsic::loongarch_lasx_xvssrlrni_d_q:
4912 case Intrinsic::loongarch_lasx_xvssrarni_d_q:
4913 case Intrinsic::loongarch_lasx_xvssrlrni_du_q:
4914 case Intrinsic::loongarch_lasx_xvssrarni_du_q:
4915 return checkIntrinsicImmArg<7>(Op, 3, DAG);
4916 case Intrinsic::loongarch_lsx_vnori_b:
4917 case Intrinsic::loongarch_lsx_vshuf4i_b:
4918 case Intrinsic::loongarch_lsx_vshuf4i_h:
4919 case Intrinsic::loongarch_lsx_vshuf4i_w:
4920 case Intrinsic::loongarch_lasx_xvnori_b:
4921 case Intrinsic::loongarch_lasx_xvshuf4i_b:
4922 case Intrinsic::loongarch_lasx_xvshuf4i_h:
4923 case Intrinsic::loongarch_lasx_xvshuf4i_w:
4924 case Intrinsic::loongarch_lasx_xvpermi_d:
4925 return checkIntrinsicImmArg<8>(Op, 2, DAG);
4926 case Intrinsic::loongarch_lsx_vshuf4i_d:
4927 case Intrinsic::loongarch_lsx_vpermi_w:
4928 case Intrinsic::loongarch_lsx_vbitseli_b:
4929 case Intrinsic::loongarch_lsx_vextrins_b:
4930 case Intrinsic::loongarch_lsx_vextrins_h:
4931 case Intrinsic::loongarch_lsx_vextrins_w:
4932 case Intrinsic::loongarch_lsx_vextrins_d:
4933 case Intrinsic::loongarch_lasx_xvshuf4i_d:
4934 case Intrinsic::loongarch_lasx_xvpermi_w:
4935 case Intrinsic::loongarch_lasx_xvpermi_q:
4936 case Intrinsic::loongarch_lasx_xvbitseli_b:
4937 case Intrinsic::loongarch_lasx_xvextrins_b:
4938 case Intrinsic::loongarch_lasx_xvextrins_h:
4939 case Intrinsic::loongarch_lasx_xvextrins_w:
4940 case Intrinsic::loongarch_lasx_xvextrins_d:
4941 return checkIntrinsicImmArg<8>(Op, 3, DAG);
4942 case Intrinsic::loongarch_lsx_vrepli_b:
4943 case Intrinsic::loongarch_lsx_vrepli_h:
4944 case Intrinsic::loongarch_lsx_vrepli_w:
4945 case Intrinsic::loongarch_lsx_vrepli_d:
4946 case Intrinsic::loongarch_lasx_xvrepli_b:
4947 case Intrinsic::loongarch_lasx_xvrepli_h:
4948 case Intrinsic::loongarch_lasx_xvrepli_w:
4949 case Intrinsic::loongarch_lasx_xvrepli_d:
4950 return checkIntrinsicImmArg<10>(Op, 1, DAG, /*IsSigned=*/true);
4951 case Intrinsic::loongarch_lsx_vldi:
4952 case Intrinsic::loongarch_lasx_xvldi:
4953 return checkIntrinsicImmArg<13>(Op, 1, DAG, /*IsSigned=*/true);
4954 }
4955}
4956
4957// Helper function that emits error message for intrinsics with chain and return
4958// merge values of a UNDEF and the chain.
4960 StringRef ErrorMsg,
4961 SelectionDAG &DAG) {
4962 DAG.getContext()->emitError(Op->getOperationName(0) + ": " + ErrorMsg + ".");
4963 return DAG.getMergeValues({DAG.getUNDEF(Op.getValueType()), Op.getOperand(0)},
4964 SDLoc(Op));
4965}
4966
4967SDValue
4968LoongArchTargetLowering::lowerINTRINSIC_W_CHAIN(SDValue Op,
4969 SelectionDAG &DAG) const {
4970 SDLoc DL(Op);
4971 MVT GRLenVT = Subtarget.getGRLenVT();
4972 EVT VT = Op.getValueType();
4973 SDValue Chain = Op.getOperand(0);
4974 const StringRef ErrorMsgOOR = "argument out of range";
4975 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
4976 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
4977
4978 switch (Op.getConstantOperandVal(1)) {
4979 default:
4980 return Op;
4981 case Intrinsic::loongarch_crc_w_b_w:
4982 case Intrinsic::loongarch_crc_w_h_w:
4983 case Intrinsic::loongarch_crc_w_w_w:
4984 case Intrinsic::loongarch_crc_w_d_w:
4985 case Intrinsic::loongarch_crcc_w_b_w:
4986 case Intrinsic::loongarch_crcc_w_h_w:
4987 case Intrinsic::loongarch_crcc_w_w_w:
4988 case Intrinsic::loongarch_crcc_w_d_w:
4989 return emitIntrinsicWithChainErrorMessage(Op, ErrorMsgReqLA64, DAG);
4990 case Intrinsic::loongarch_csrrd_w:
4991 case Intrinsic::loongarch_csrrd_d: {
4992 unsigned Imm = Op.getConstantOperandVal(2);
4993 return !isUInt<14>(Imm)
4994 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
4995 : DAG.getNode(LoongArchISD::CSRRD, DL, {GRLenVT, MVT::Other},
4996 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
4997 }
4998 case Intrinsic::loongarch_csrwr_w:
4999 case Intrinsic::loongarch_csrwr_d: {
5000 unsigned Imm = Op.getConstantOperandVal(3);
5001 return !isUInt<14>(Imm)
5002 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5003 : DAG.getNode(LoongArchISD::CSRWR, DL, {GRLenVT, MVT::Other},
5004 {Chain, Op.getOperand(2),
5005 DAG.getConstant(Imm, DL, GRLenVT)});
5006 }
5007 case Intrinsic::loongarch_csrxchg_w:
5008 case Intrinsic::loongarch_csrxchg_d: {
5009 unsigned Imm = Op.getConstantOperandVal(4);
5010 return !isUInt<14>(Imm)
5011 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5012 : DAG.getNode(LoongArchISD::CSRXCHG, DL, {GRLenVT, MVT::Other},
5013 {Chain, Op.getOperand(2), Op.getOperand(3),
5014 DAG.getConstant(Imm, DL, GRLenVT)});
5015 }
5016 case Intrinsic::loongarch_iocsrrd_d: {
5017 return DAG.getNode(
5018 LoongArchISD::IOCSRRD_D, DL, {GRLenVT, MVT::Other},
5019 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op.getOperand(2))});
5020 }
5021#define IOCSRRD_CASE(NAME, NODE) \
5022 case Intrinsic::loongarch_##NAME: { \
5023 return DAG.getNode(LoongArchISD::NODE, DL, {GRLenVT, MVT::Other}, \
5024 {Chain, Op.getOperand(2)}); \
5025 }
5026 IOCSRRD_CASE(iocsrrd_b, IOCSRRD_B);
5027 IOCSRRD_CASE(iocsrrd_h, IOCSRRD_H);
5028 IOCSRRD_CASE(iocsrrd_w, IOCSRRD_W);
5029#undef IOCSRRD_CASE
5030 case Intrinsic::loongarch_cpucfg: {
5031 return DAG.getNode(LoongArchISD::CPUCFG, DL, {GRLenVT, MVT::Other},
5032 {Chain, Op.getOperand(2)});
5033 }
5034 case Intrinsic::loongarch_lddir_d: {
5035 unsigned Imm = Op.getConstantOperandVal(3);
5036 return !isUInt<8>(Imm)
5037 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5038 : Op;
5039 }
5040 case Intrinsic::loongarch_movfcsr2gr: {
5041 if (!Subtarget.hasBasicF())
5042 return emitIntrinsicWithChainErrorMessage(Op, ErrorMsgReqF, DAG);
5043 unsigned Imm = Op.getConstantOperandVal(2);
5044 return !isUInt<2>(Imm)
5045 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5046 : DAG.getNode(LoongArchISD::MOVFCSR2GR, DL, {VT, MVT::Other},
5047 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5048 }
5049 case Intrinsic::loongarch_lsx_vld:
5050 case Intrinsic::loongarch_lsx_vldrepl_b:
5051 case Intrinsic::loongarch_lasx_xvld:
5052 case Intrinsic::loongarch_lasx_xvldrepl_b:
5053 return !isInt<12>(cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5054 ? emitIntrinsicWithChainErrorMessage(Op, ErrorMsgOOR, DAG)
5055 : SDValue();
5056 case Intrinsic::loongarch_lsx_vldrepl_h:
5057 case Intrinsic::loongarch_lasx_xvldrepl_h:
5058 return !isShiftedInt<11, 1>(
5059 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5061 Op, "argument out of range or not a multiple of 2", DAG)
5062 : SDValue();
5063 case Intrinsic::loongarch_lsx_vldrepl_w:
5064 case Intrinsic::loongarch_lasx_xvldrepl_w:
5065 return !isShiftedInt<10, 2>(
5066 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5068 Op, "argument out of range or not a multiple of 4", DAG)
5069 : SDValue();
5070 case Intrinsic::loongarch_lsx_vldrepl_d:
5071 case Intrinsic::loongarch_lasx_xvldrepl_d:
5072 return !isShiftedInt<9, 3>(
5073 cast<ConstantSDNode>(Op.getOperand(3))->getSExtValue())
5075 Op, "argument out of range or not a multiple of 8", DAG)
5076 : SDValue();
5077 }
5078}
5079
5080// Helper function that emits error message for intrinsics with void return
5081// value and return the chain.
5083 SelectionDAG &DAG) {
5084
5085 DAG.getContext()->emitError(Op->getOperationName(0) + ": " + ErrorMsg + ".");
5086 return Op.getOperand(0);
5087}
5088
5089SDValue LoongArchTargetLowering::lowerINTRINSIC_VOID(SDValue Op,
5090 SelectionDAG &DAG) const {
5091 SDLoc DL(Op);
5092 MVT GRLenVT = Subtarget.getGRLenVT();
5093 SDValue Chain = Op.getOperand(0);
5094 uint64_t IntrinsicEnum = Op.getConstantOperandVal(1);
5095 SDValue Op2 = Op.getOperand(2);
5096 const StringRef ErrorMsgOOR = "argument out of range";
5097 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
5098 const StringRef ErrorMsgReqLA32 = "requires loongarch32";
5099 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
5100
5101 switch (IntrinsicEnum) {
5102 default:
5103 // TODO: Add more Intrinsics.
5104 return SDValue();
5105 case Intrinsic::loongarch_cacop_d:
5106 case Intrinsic::loongarch_cacop_w: {
5107 if (IntrinsicEnum == Intrinsic::loongarch_cacop_d && !Subtarget.is64Bit())
5108 return emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG);
5109 if (IntrinsicEnum == Intrinsic::loongarch_cacop_w && Subtarget.is64Bit())
5110 return emitIntrinsicErrorMessage(Op, ErrorMsgReqLA32, DAG);
5111 // call void @llvm.loongarch.cacop.[d/w](uimm5, rj, simm12)
5112 unsigned Imm1 = Op2->getAsZExtVal();
5113 int Imm2 = cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue();
5114 if (!isUInt<5>(Imm1) || !isInt<12>(Imm2))
5115 return emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG);
5116 return Op;
5117 }
5118 case Intrinsic::loongarch_dbar: {
5119 unsigned Imm = Op2->getAsZExtVal();
5120 return !isUInt<15>(Imm)
5121 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5122 : DAG.getNode(LoongArchISD::DBAR, DL, MVT::Other, Chain,
5123 DAG.getConstant(Imm, DL, GRLenVT));
5124 }
5125 case Intrinsic::loongarch_ibar: {
5126 unsigned Imm = Op2->getAsZExtVal();
5127 return !isUInt<15>(Imm)
5128 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5129 : DAG.getNode(LoongArchISD::IBAR, DL, MVT::Other, Chain,
5130 DAG.getConstant(Imm, DL, GRLenVT));
5131 }
5132 case Intrinsic::loongarch_break: {
5133 unsigned Imm = Op2->getAsZExtVal();
5134 return !isUInt<15>(Imm)
5135 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5136 : DAG.getNode(LoongArchISD::BREAK, DL, MVT::Other, Chain,
5137 DAG.getConstant(Imm, DL, GRLenVT));
5138 }
5139 case Intrinsic::loongarch_movgr2fcsr: {
5140 if (!Subtarget.hasBasicF())
5141 return emitIntrinsicErrorMessage(Op, ErrorMsgReqF, DAG);
5142 unsigned Imm = Op2->getAsZExtVal();
5143 return !isUInt<2>(Imm)
5144 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5145 : DAG.getNode(LoongArchISD::MOVGR2FCSR, DL, MVT::Other, Chain,
5146 DAG.getConstant(Imm, DL, GRLenVT),
5147 DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT,
5148 Op.getOperand(3)));
5149 }
5150 case Intrinsic::loongarch_syscall: {
5151 unsigned Imm = Op2->getAsZExtVal();
5152 return !isUInt<15>(Imm)
5153 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5154 : DAG.getNode(LoongArchISD::SYSCALL, DL, MVT::Other, Chain,
5155 DAG.getConstant(Imm, DL, GRLenVT));
5156 }
5157#define IOCSRWR_CASE(NAME, NODE) \
5158 case Intrinsic::loongarch_##NAME: { \
5159 SDValue Op3 = Op.getOperand(3); \
5160 return Subtarget.is64Bit() \
5161 ? DAG.getNode(LoongArchISD::NODE, DL, MVT::Other, Chain, \
5162 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2), \
5163 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op3)) \
5164 : DAG.getNode(LoongArchISD::NODE, DL, MVT::Other, Chain, Op2, \
5165 Op3); \
5166 }
5167 IOCSRWR_CASE(iocsrwr_b, IOCSRWR_B);
5168 IOCSRWR_CASE(iocsrwr_h, IOCSRWR_H);
5169 IOCSRWR_CASE(iocsrwr_w, IOCSRWR_W);
5170#undef IOCSRWR_CASE
5171 case Intrinsic::loongarch_iocsrwr_d: {
5172 return !Subtarget.is64Bit()
5173 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG)
5174 : DAG.getNode(LoongArchISD::IOCSRWR_D, DL, MVT::Other, Chain,
5175 Op2,
5176 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64,
5177 Op.getOperand(3)));
5178 }
5179#define ASRT_LE_GT_CASE(NAME) \
5180 case Intrinsic::loongarch_##NAME: { \
5181 return !Subtarget.is64Bit() \
5182 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG) \
5183 : Op; \
5184 }
5185 ASRT_LE_GT_CASE(asrtle_d)
5186 ASRT_LE_GT_CASE(asrtgt_d)
5187#undef ASRT_LE_GT_CASE
5188 case Intrinsic::loongarch_ldpte_d: {
5189 unsigned Imm = Op.getConstantOperandVal(3);
5190 return !Subtarget.is64Bit()
5191 ? emitIntrinsicErrorMessage(Op, ErrorMsgReqLA64, DAG)
5192 : !isUInt<8>(Imm) ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5193 : Op;
5194 }
5195 case Intrinsic::loongarch_lsx_vst:
5196 case Intrinsic::loongarch_lasx_xvst:
5197 return !isInt<12>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue())
5198 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5199 : SDValue();
5200 case Intrinsic::loongarch_lasx_xvstelm_b:
5201 return (!isInt<8>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5202 !isUInt<5>(Op.getConstantOperandVal(5)))
5203 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5204 : SDValue();
5205 case Intrinsic::loongarch_lsx_vstelm_b:
5206 return (!isInt<8>(cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5207 !isUInt<4>(Op.getConstantOperandVal(5)))
5208 ? emitIntrinsicErrorMessage(Op, ErrorMsgOOR, DAG)
5209 : SDValue();
5210 case Intrinsic::loongarch_lasx_xvstelm_h:
5211 return (!isShiftedInt<8, 1>(
5212 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5213 !isUInt<4>(Op.getConstantOperandVal(5)))
5215 Op, "argument out of range or not a multiple of 2", DAG)
5216 : SDValue();
5217 case Intrinsic::loongarch_lsx_vstelm_h:
5218 return (!isShiftedInt<8, 1>(
5219 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5220 !isUInt<3>(Op.getConstantOperandVal(5)))
5222 Op, "argument out of range or not a multiple of 2", DAG)
5223 : SDValue();
5224 case Intrinsic::loongarch_lasx_xvstelm_w:
5225 return (!isShiftedInt<8, 2>(
5226 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5227 !isUInt<3>(Op.getConstantOperandVal(5)))
5229 Op, "argument out of range or not a multiple of 4", DAG)
5230 : SDValue();
5231 case Intrinsic::loongarch_lsx_vstelm_w:
5232 return (!isShiftedInt<8, 2>(
5233 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5234 !isUInt<2>(Op.getConstantOperandVal(5)))
5236 Op, "argument out of range or not a multiple of 4", DAG)
5237 : SDValue();
5238 case Intrinsic::loongarch_lasx_xvstelm_d:
5239 return (!isShiftedInt<8, 3>(
5240 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5241 !isUInt<2>(Op.getConstantOperandVal(5)))
5243 Op, "argument out of range or not a multiple of 8", DAG)
5244 : SDValue();
5245 case Intrinsic::loongarch_lsx_vstelm_d:
5246 return (!isShiftedInt<8, 3>(
5247 cast<ConstantSDNode>(Op.getOperand(4))->getSExtValue()) ||
5248 !isUInt<1>(Op.getConstantOperandVal(5)))
5250 Op, "argument out of range or not a multiple of 8", DAG)
5251 : SDValue();
5252 }
5253}
5254
5255SDValue LoongArchTargetLowering::lowerShiftLeftParts(SDValue Op,
5256 SelectionDAG &DAG) const {
5257 SDLoc DL(Op);
5258 SDValue Lo = Op.getOperand(0);
5259 SDValue Hi = Op.getOperand(1);
5260 SDValue Shamt = Op.getOperand(2);
5261 EVT VT = Lo.getValueType();
5262
5263 // if Shamt-GRLen < 0: // Shamt < GRLen
5264 // Lo = Lo << Shamt
5265 // Hi = (Hi << Shamt) | ((Lo >>u 1) >>u (GRLen-1 ^ Shamt))
5266 // else:
5267 // Lo = 0
5268 // Hi = Lo << (Shamt-GRLen)
5269
5270 SDValue Zero = DAG.getConstant(0, DL, VT);
5271 SDValue One = DAG.getConstant(1, DL, VT);
5272 SDValue MinusGRLen =
5273 DAG.getSignedConstant(-(int)Subtarget.getGRLen(), DL, VT);
5274 SDValue GRLenMinus1 = DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT);
5275 SDValue ShamtMinusGRLen = DAG.getNode(ISD::ADD, DL, VT, Shamt, MinusGRLen);
5276 SDValue GRLenMinus1Shamt = DAG.getNode(ISD::XOR, DL, VT, Shamt, GRLenMinus1);
5277
5278 SDValue LoTrue = DAG.getNode(ISD::SHL, DL, VT, Lo, Shamt);
5279 SDValue ShiftRight1Lo = DAG.getNode(ISD::SRL, DL, VT, Lo, One);
5280 SDValue ShiftRightLo =
5281 DAG.getNode(ISD::SRL, DL, VT, ShiftRight1Lo, GRLenMinus1Shamt);
5282 SDValue ShiftLeftHi = DAG.getNode(ISD::SHL, DL, VT, Hi, Shamt);
5283 SDValue HiTrue = DAG.getNode(ISD::OR, DL, VT, ShiftLeftHi, ShiftRightLo);
5284 SDValue HiFalse = DAG.getNode(ISD::SHL, DL, VT, Lo, ShamtMinusGRLen);
5285
5286 SDValue CC = DAG.getSetCC(DL, VT, ShamtMinusGRLen, Zero, ISD::SETLT);
5287
5288 Lo = DAG.getNode(ISD::SELECT, DL, VT, CC, LoTrue, Zero);
5289 Hi = DAG.getNode(ISD::SELECT, DL, VT, CC, HiTrue, HiFalse);
5290
5291 SDValue Parts[2] = {Lo, Hi};
5292 return DAG.getMergeValues(Parts, DL);
5293}
5294
5295SDValue LoongArchTargetLowering::lowerShiftRightParts(SDValue Op,
5296 SelectionDAG &DAG,
5297 bool IsSRA) const {
5298 SDLoc DL(Op);
5299 SDValue Lo = Op.getOperand(0);
5300 SDValue Hi = Op.getOperand(1);
5301 SDValue Shamt = Op.getOperand(2);
5302 EVT VT = Lo.getValueType();
5303
5304 // SRA expansion:
5305 // if Shamt-GRLen < 0: // Shamt < GRLen
5306 // Lo = (Lo >>u Shamt) | ((Hi << 1) << (ShAmt ^ GRLen-1))
5307 // Hi = Hi >>s Shamt
5308 // else:
5309 // Lo = Hi >>s (Shamt-GRLen);
5310 // Hi = Hi >>s (GRLen-1)
5311 //
5312 // SRL expansion:
5313 // if Shamt-GRLen < 0: // Shamt < GRLen
5314 // Lo = (Lo >>u Shamt) | ((Hi << 1) << (ShAmt ^ GRLen-1))
5315 // Hi = Hi >>u Shamt
5316 // else:
5317 // Lo = Hi >>u (Shamt-GRLen);
5318 // Hi = 0;
5319
5320 unsigned ShiftRightOp = IsSRA ? ISD::SRA : ISD::SRL;
5321
5322 SDValue Zero = DAG.getConstant(0, DL, VT);
5323 SDValue One = DAG.getConstant(1, DL, VT);
5324 SDValue MinusGRLen =
5325 DAG.getSignedConstant(-(int)Subtarget.getGRLen(), DL, VT);
5326 SDValue GRLenMinus1 = DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT);
5327 SDValue ShamtMinusGRLen = DAG.getNode(ISD::ADD, DL, VT, Shamt, MinusGRLen);
5328 SDValue GRLenMinus1Shamt = DAG.getNode(ISD::XOR, DL, VT, Shamt, GRLenMinus1);
5329
5330 SDValue ShiftRightLo = DAG.getNode(ISD::SRL, DL, VT, Lo, Shamt);
5331 SDValue ShiftLeftHi1 = DAG.getNode(ISD::SHL, DL, VT, Hi, One);
5332 SDValue ShiftLeftHi =
5333 DAG.getNode(ISD::SHL, DL, VT, ShiftLeftHi1, GRLenMinus1Shamt);
5334 SDValue LoTrue = DAG.getNode(ISD::OR, DL, VT, ShiftRightLo, ShiftLeftHi);
5335 SDValue HiTrue = DAG.getNode(ShiftRightOp, DL, VT, Hi, Shamt);
5336 SDValue LoFalse = DAG.getNode(ShiftRightOp, DL, VT, Hi, ShamtMinusGRLen);
5337 SDValue HiFalse =
5338 IsSRA ? DAG.getNode(ISD::SRA, DL, VT, Hi, GRLenMinus1) : Zero;
5339
5340 SDValue CC = DAG.getSetCC(DL, VT, ShamtMinusGRLen, Zero, ISD::SETLT);
5341
5342 Lo = DAG.getNode(ISD::SELECT, DL, VT, CC, LoTrue, LoFalse);
5343 Hi = DAG.getNode(ISD::SELECT, DL, VT, CC, HiTrue, HiFalse);
5344
5345 SDValue Parts[2] = {Lo, Hi};
5346 return DAG.getMergeValues(Parts, DL);
5347}
5348
5349// Returns the opcode of the target-specific SDNode that implements the 32-bit
5350// form of the given Opcode.
5351static unsigned getLoongArchWOpcode(unsigned Opcode) {
5352 switch (Opcode) {
5353 default:
5354 llvm_unreachable("Unexpected opcode");
5355 case ISD::SDIV:
5356 return LoongArchISD::DIV_W;
5357 case ISD::UDIV:
5358 return LoongArchISD::DIV_WU;
5359 case ISD::SREM:
5360 return LoongArchISD::MOD_W;
5361 case ISD::UREM:
5362 return LoongArchISD::MOD_WU;
5363 case ISD::SHL:
5364 return LoongArchISD::SLL_W;
5365 case ISD::SRA:
5366 return LoongArchISD::SRA_W;
5367 case ISD::SRL:
5368 return LoongArchISD::SRL_W;
5369 case ISD::ROTL:
5370 case ISD::ROTR:
5371 return LoongArchISD::ROTR_W;
5372 case ISD::CTTZ:
5373 return LoongArchISD::CTZ_W;
5374 case ISD::CTLZ:
5375 return LoongArchISD::CLZ_W;
5376 }
5377}
5378
5379// Converts the given i8/i16/i32 operation to a target-specific SelectionDAG
5380// node. Because i8/i16/i32 isn't a legal type for LA64, these operations would
5381// otherwise be promoted to i64, making it difficult to select the
5382// SLL_W/.../*W later one because the fact the operation was originally of
5383// type i8/i16/i32 is lost.
5385 unsigned ExtOpc = ISD::ANY_EXTEND) {
5386 SDLoc DL(N);
5387 unsigned WOpcode = getLoongArchWOpcode(N->getOpcode());
5388 SDValue NewOp0, NewRes;
5389
5390 switch (NumOp) {
5391 default:
5392 llvm_unreachable("Unexpected NumOp");
5393 case 1: {
5394 NewOp0 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(0));
5395 NewRes = DAG.getNode(WOpcode, DL, MVT::i64, NewOp0);
5396 break;
5397 }
5398 case 2: {
5399 NewOp0 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(0));
5400 SDValue NewOp1 = DAG.getNode(ExtOpc, DL, MVT::i64, N->getOperand(1));
5401 if (N->getOpcode() == ISD::ROTL) {
5402 SDValue TmpOp = DAG.getConstant(32, DL, MVT::i64);
5403 NewOp1 = DAG.getNode(ISD::SUB, DL, MVT::i64, TmpOp, NewOp1);
5404 }
5405 NewRes = DAG.getNode(WOpcode, DL, MVT::i64, NewOp0, NewOp1);
5406 break;
5407 }
5408 // TODO:Handle more NumOp.
5409 }
5410
5411 // ReplaceNodeResults requires we maintain the same type for the return
5412 // value.
5413 return DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), NewRes);
5414}
5415
5416// Converts the given 32-bit operation to a i64 operation with signed extension
5417// semantic to reduce the signed extension instructions.
5419 SDLoc DL(N);
5420 SDValue NewOp0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(0));
5421 SDValue NewOp1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(1));
5422 SDValue NewWOp = DAG.getNode(N->getOpcode(), DL, MVT::i64, NewOp0, NewOp1);
5423 SDValue NewRes = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i64, NewWOp,
5424 DAG.getValueType(MVT::i32));
5425 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, NewRes);
5426}
5427
5428// Helper function that emits error message for intrinsics with/without chain
5429// and return a UNDEF or and the chain as the results.
5432 StringRef ErrorMsg, bool WithChain = true) {
5433 DAG.getContext()->emitError(N->getOperationName(0) + ": " + ErrorMsg + ".");
5434 Results.push_back(DAG.getUNDEF(N->getValueType(0)));
5435 if (!WithChain)
5436 return;
5437 Results.push_back(N->getOperand(0));
5438}
5439
5440template <unsigned N>
5441static void
5443 SelectionDAG &DAG, const LoongArchSubtarget &Subtarget,
5444 unsigned ResOp) {
5445 const StringRef ErrorMsgOOR = "argument out of range";
5446 unsigned Imm = Node->getConstantOperandVal(2);
5447 if (!isUInt<N>(Imm)) {
5449 /*WithChain=*/false);
5450 return;
5451 }
5452 SDLoc DL(Node);
5453 SDValue Vec = Node->getOperand(1);
5454
5455 SDValue PickElt =
5456 DAG.getNode(ResOp, DL, Subtarget.getGRLenVT(), Vec,
5457 DAG.getConstant(Imm, DL, Subtarget.getGRLenVT()),
5459 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, Node->getValueType(0),
5460 PickElt.getValue(0)));
5461}
5462
5465 SelectionDAG &DAG,
5466 const LoongArchSubtarget &Subtarget,
5467 unsigned ResOp) {
5468 SDLoc DL(N);
5469 SDValue Vec = N->getOperand(1);
5470
5471 SDValue CB = DAG.getNode(ResOp, DL, Subtarget.getGRLenVT(), Vec);
5472 Results.push_back(
5473 DAG.getNode(ISD::TRUNCATE, DL, N->getValueType(0), CB.getValue(0)));
5474}
5475
5476static void
5478 SelectionDAG &DAG,
5479 const LoongArchSubtarget &Subtarget) {
5480 switch (N->getConstantOperandVal(0)) {
5481 default:
5482 llvm_unreachable("Unexpected Intrinsic.");
5483 case Intrinsic::loongarch_lsx_vpickve2gr_b:
5484 replaceVPICKVE2GRResults<4>(N, Results, DAG, Subtarget,
5485 LoongArchISD::VPICK_SEXT_ELT);
5486 break;
5487 case Intrinsic::loongarch_lsx_vpickve2gr_h:
5488 case Intrinsic::loongarch_lasx_xvpickve2gr_w:
5489 replaceVPICKVE2GRResults<3>(N, Results, DAG, Subtarget,
5490 LoongArchISD::VPICK_SEXT_ELT);
5491 break;
5492 case Intrinsic::loongarch_lsx_vpickve2gr_w:
5493 replaceVPICKVE2GRResults<2>(N, Results, DAG, Subtarget,
5494 LoongArchISD::VPICK_SEXT_ELT);
5495 break;
5496 case Intrinsic::loongarch_lsx_vpickve2gr_bu:
5497 replaceVPICKVE2GRResults<4>(N, Results, DAG, Subtarget,
5498 LoongArchISD::VPICK_ZEXT_ELT);
5499 break;
5500 case Intrinsic::loongarch_lsx_vpickve2gr_hu:
5501 case Intrinsic::loongarch_lasx_xvpickve2gr_wu:
5502 replaceVPICKVE2GRResults<3>(N, Results, DAG, Subtarget,
5503 LoongArchISD::VPICK_ZEXT_ELT);
5504 break;
5505 case Intrinsic::loongarch_lsx_vpickve2gr_wu:
5506 replaceVPICKVE2GRResults<2>(N, Results, DAG, Subtarget,
5507 LoongArchISD::VPICK_ZEXT_ELT);
5508 break;
5509 case Intrinsic::loongarch_lsx_bz_b:
5510 case Intrinsic::loongarch_lsx_bz_h:
5511 case Intrinsic::loongarch_lsx_bz_w:
5512 case Intrinsic::loongarch_lsx_bz_d:
5513 case Intrinsic::loongarch_lasx_xbz_b:
5514 case Intrinsic::loongarch_lasx_xbz_h:
5515 case Intrinsic::loongarch_lasx_xbz_w:
5516 case Intrinsic::loongarch_lasx_xbz_d:
5517 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5518 LoongArchISD::VALL_ZERO);
5519 break;
5520 case Intrinsic::loongarch_lsx_bz_v:
5521 case Intrinsic::loongarch_lasx_xbz_v:
5522 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5523 LoongArchISD::VANY_ZERO);
5524 break;
5525 case Intrinsic::loongarch_lsx_bnz_b:
5526 case Intrinsic::loongarch_lsx_bnz_h:
5527 case Intrinsic::loongarch_lsx_bnz_w:
5528 case Intrinsic::loongarch_lsx_bnz_d:
5529 case Intrinsic::loongarch_lasx_xbnz_b:
5530 case Intrinsic::loongarch_lasx_xbnz_h:
5531 case Intrinsic::loongarch_lasx_xbnz_w:
5532 case Intrinsic::loongarch_lasx_xbnz_d:
5533 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5534 LoongArchISD::VALL_NONZERO);
5535 break;
5536 case Intrinsic::loongarch_lsx_bnz_v:
5537 case Intrinsic::loongarch_lasx_xbnz_v:
5538 replaceVecCondBranchResults(N, Results, DAG, Subtarget,
5539 LoongArchISD::VANY_NONZERO);
5540 break;
5541 }
5542}
5543
5546 SelectionDAG &DAG) {
5547 assert(N->getValueType(0) == MVT::i128 &&
5548 "AtomicCmpSwap on types less than 128 should be legal");
5549 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
5550
5551 unsigned Opcode;
5552 switch (MemOp->getMergedOrdering()) {
5556 Opcode = LoongArch::PseudoCmpXchg128Acquire;
5557 break;
5560 Opcode = LoongArch::PseudoCmpXchg128;
5561 break;
5562 default:
5563 llvm_unreachable("Unexpected ordering!");
5564 }
5565
5566 SDLoc DL(N);
5567 auto CmpVal = DAG.SplitScalar(N->getOperand(2), DL, MVT::i64, MVT::i64);
5568 auto NewVal = DAG.SplitScalar(N->getOperand(3), DL, MVT::i64, MVT::i64);
5569 SDValue Ops[] = {N->getOperand(1), CmpVal.first, CmpVal.second,
5570 NewVal.first, NewVal.second, N->getOperand(0)};
5571
5572 SDNode *CmpSwap = DAG.getMachineNode(
5573 Opcode, SDLoc(N), DAG.getVTList(MVT::i64, MVT::i64, MVT::i64, MVT::Other),
5574 Ops);
5575 DAG.setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp});
5576 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i128,
5577 SDValue(CmpSwap, 0), SDValue(CmpSwap, 1)));
5578 Results.push_back(SDValue(CmpSwap, 3));
5579}
5580
5583 SDLoc DL(N);
5584 EVT VT = N->getValueType(0);
5585 switch (N->getOpcode()) {
5586 default:
5587 llvm_unreachable("Don't know how to legalize this operation");
5588 case ISD::ADD:
5589 case ISD::SUB:
5590 assert(N->getValueType(0) == MVT::i32 && Subtarget.is64Bit() &&
5591 "Unexpected custom legalisation");
5592 Results.push_back(customLegalizeToWOpWithSExt(N, DAG));
5593 break;
5594 case ISD::SDIV:
5595 case ISD::UDIV:
5596 case ISD::SREM:
5597 case ISD::UREM:
5598 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5599 "Unexpected custom legalisation");
5600 Results.push_back(customLegalizeToWOp(N, DAG, 2,
5601 Subtarget.hasDiv32() && VT == MVT::i32
5603 : ISD::SIGN_EXTEND));
5604 break;
5605 case ISD::SHL:
5606 case ISD::SRA:
5607 case ISD::SRL:
5608 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5609 "Unexpected custom legalisation");
5610 if (N->getOperand(1).getOpcode() != ISD::Constant) {
5611 Results.push_back(customLegalizeToWOp(N, DAG, 2));
5612 break;
5613 }
5614 break;
5615 case ISD::ROTL:
5616 case ISD::ROTR:
5617 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5618 "Unexpected custom legalisation");
5619 Results.push_back(customLegalizeToWOp(N, DAG, 2));
5620 break;
5621 case ISD::LOAD: {
5622 // Use an f64 load and a scalar_to_vector for v2f32 loads. This avoids
5623 // scalarizing in 32-bit mode. In 64-bit mode this avoids a int->fp
5624 // cast since type legalization will try to use an i64 load.
5625 MVT VT = N->getSimpleValueType(0);
5626 assert(VT == MVT::v2f32 && Subtarget.hasExtLSX() &&
5627 "Unexpected custom legalisation");
5629 "Unexpected type action!");
5630 if (!ISD::isNON_EXTLoad(N))
5631 return;
5632 auto *Ld = cast<LoadSDNode>(N);
5633 SDValue Res = DAG.getLoad(MVT::f64, DL, Ld->getChain(), Ld->getBasePtr(),
5634 Ld->getPointerInfo(), Ld->getBaseAlign(),
5635 Ld->getMemOperand()->getFlags());
5636 SDValue Chain = Res.getValue(1);
5637 MVT VecVT = MVT::getVectorVT(MVT::f64, 2);
5638 Res = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VecVT, Res);
5639 EVT WideVT = getTypeToTransformTo(*DAG.getContext(), VT);
5640 Res = DAG.getBitcast(WideVT, Res);
5641 Results.push_back(Res);
5642 Results.push_back(Chain);
5643 break;
5644 }
5645 case ISD::FP_TO_SINT: {
5646 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5647 "Unexpected custom legalisation");
5648 SDValue Src = N->getOperand(0);
5649 EVT FVT = EVT::getFloatingPointVT(N->getValueSizeInBits(0));
5650 if (getTypeAction(*DAG.getContext(), Src.getValueType()) !=
5652 if (!isTypeLegal(Src.getValueType()))
5653 return;
5654 if (Src.getValueType() == MVT::f16)
5655 Src = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
5656 SDValue Dst = DAG.getNode(LoongArchISD::FTINT, DL, FVT, Src);
5657 Results.push_back(DAG.getNode(ISD::BITCAST, DL, VT, Dst));
5658 return;
5659 }
5660 // If the FP type needs to be softened, emit a library call using the 'si'
5661 // version. If we left it to default legalization we'd end up with 'di'.
5662 RTLIB::Libcall LC;
5663 LC = RTLIB::getFPTOSINT(Src.getValueType(), VT);
5664 MakeLibCallOptions CallOptions;
5665 EVT OpVT = Src.getValueType();
5666 CallOptions.setTypeListBeforeSoften(OpVT, VT);
5667 SDValue Chain = SDValue();
5668 SDValue Result;
5669 std::tie(Result, Chain) =
5670 makeLibCall(DAG, LC, VT, Src, CallOptions, DL, Chain);
5671 Results.push_back(Result);
5672 break;
5673 }
5674 case ISD::BITCAST: {
5675 SDValue Src = N->getOperand(0);
5676 EVT SrcVT = Src.getValueType();
5677 if (VT == MVT::i32 && SrcVT == MVT::f32 && Subtarget.is64Bit() &&
5678 Subtarget.hasBasicF()) {
5679 SDValue Dst =
5680 DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Src);
5681 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Dst));
5682 } else if (VT == MVT::i64 && SrcVT == MVT::f64 && !Subtarget.is64Bit()) {
5683 SDValue NewReg = DAG.getNode(LoongArchISD::SPLIT_PAIR_F64, DL,
5684 DAG.getVTList(MVT::i32, MVT::i32), Src);
5685 SDValue RetReg = DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64,
5686 NewReg.getValue(0), NewReg.getValue(1));
5687 Results.push_back(RetReg);
5688 }
5689 break;
5690 }
5691 case ISD::FP_TO_UINT: {
5692 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5693 "Unexpected custom legalisation");
5694 auto &TLI = DAG.getTargetLoweringInfo();
5695 SDValue Tmp1, Tmp2;
5696 TLI.expandFP_TO_UINT(N, Tmp1, Tmp2, DAG);
5697 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Tmp1));
5698 break;
5699 }
5700 case ISD::FP_ROUND: {
5701 assert(VT == MVT::v2f32 && Subtarget.hasExtLSX() &&
5702 "Unexpected custom legalisation");
5703 // On LSX platforms, rounding from v2f64 to v4f32 (after legalization from
5704 // v2f32) is scalarized. Add a customized v2f32 widening to convert it into
5705 // a target-specific LoongArchISD::VFCVT to optimize it.
5706 SDValue Op0 = N->getOperand(0);
5707 EVT OpVT = Op0.getValueType();
5708 if (OpVT == MVT::v2f64) {
5709 SDValue Undef = DAG.getUNDEF(OpVT);
5710 SDValue Dst =
5711 DAG.getNode(LoongArchISD::VFCVT, DL, MVT::v4f32, Undef, Op0);
5712 Results.push_back(Dst);
5713 }
5714 break;
5715 }
5716 case ISD::BSWAP: {
5717 SDValue Src = N->getOperand(0);
5718 assert((VT == MVT::i16 || VT == MVT::i32) &&
5719 "Unexpected custom legalization");
5720 MVT GRLenVT = Subtarget.getGRLenVT();
5721 SDValue NewSrc = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, Src);
5722 SDValue Tmp;
5723 switch (VT.getSizeInBits()) {
5724 default:
5725 llvm_unreachable("Unexpected operand width");
5726 case 16:
5727 Tmp = DAG.getNode(LoongArchISD::REVB_2H, DL, GRLenVT, NewSrc);
5728 break;
5729 case 32:
5730 // Only LA64 will get to here due to the size mismatch between VT and
5731 // GRLenVT, LA32 lowering is directly defined in LoongArchInstrInfo.
5732 Tmp = DAG.getNode(LoongArchISD::REVB_2W, DL, GRLenVT, NewSrc);
5733 break;
5734 }
5735 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Tmp));
5736 break;
5737 }
5738 case ISD::BITREVERSE: {
5739 SDValue Src = N->getOperand(0);
5740 assert((VT == MVT::i8 || (VT == MVT::i32 && Subtarget.is64Bit())) &&
5741 "Unexpected custom legalization");
5742 MVT GRLenVT = Subtarget.getGRLenVT();
5743 SDValue NewSrc = DAG.getNode(ISD::ANY_EXTEND, DL, GRLenVT, Src);
5744 SDValue Tmp;
5745 switch (VT.getSizeInBits()) {
5746 default:
5747 llvm_unreachable("Unexpected operand width");
5748 case 8:
5749 Tmp = DAG.getNode(LoongArchISD::BITREV_4B, DL, GRLenVT, NewSrc);
5750 break;
5751 case 32:
5752 Tmp = DAG.getNode(LoongArchISD::BITREV_W, DL, GRLenVT, NewSrc);
5753 break;
5754 }
5755 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, Tmp));
5756 break;
5757 }
5758 case ISD::CTLZ:
5759 case ISD::CTTZ: {
5760 assert(VT == MVT::i32 && Subtarget.is64Bit() &&
5761 "Unexpected custom legalisation");
5762 Results.push_back(customLegalizeToWOp(N, DAG, 1));
5763 break;
5764 }
5766 SDValue Chain = N->getOperand(0);
5767 SDValue Op2 = N->getOperand(2);
5768 MVT GRLenVT = Subtarget.getGRLenVT();
5769 const StringRef ErrorMsgOOR = "argument out of range";
5770 const StringRef ErrorMsgReqLA64 = "requires loongarch64";
5771 const StringRef ErrorMsgReqF = "requires basic 'f' target feature";
5772
5773 switch (N->getConstantOperandVal(1)) {
5774 default:
5775 llvm_unreachable("Unexpected Intrinsic.");
5776 case Intrinsic::loongarch_movfcsr2gr: {
5777 if (!Subtarget.hasBasicF()) {
5778 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqF);
5779 return;
5780 }
5781 unsigned Imm = Op2->getAsZExtVal();
5782 if (!isUInt<2>(Imm)) {
5783 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5784 return;
5785 }
5786 SDValue MOVFCSR2GRResults = DAG.getNode(
5787 LoongArchISD::MOVFCSR2GR, SDLoc(N), {MVT::i64, MVT::Other},
5788 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5789 Results.push_back(
5790 DAG.getNode(ISD::TRUNCATE, DL, VT, MOVFCSR2GRResults.getValue(0)));
5791 Results.push_back(MOVFCSR2GRResults.getValue(1));
5792 break;
5793 }
5794#define CRC_CASE_EXT_BINARYOP(NAME, NODE) \
5795 case Intrinsic::loongarch_##NAME: { \
5796 SDValue NODE = DAG.getNode( \
5797 LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5798 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2), \
5799 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3))}); \
5800 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, NODE.getValue(0))); \
5801 Results.push_back(NODE.getValue(1)); \
5802 break; \
5803 }
5804 CRC_CASE_EXT_BINARYOP(crc_w_b_w, CRC_W_B_W)
5805 CRC_CASE_EXT_BINARYOP(crc_w_h_w, CRC_W_H_W)
5806 CRC_CASE_EXT_BINARYOP(crc_w_w_w, CRC_W_W_W)
5807 CRC_CASE_EXT_BINARYOP(crcc_w_b_w, CRCC_W_B_W)
5808 CRC_CASE_EXT_BINARYOP(crcc_w_h_w, CRCC_W_H_W)
5809 CRC_CASE_EXT_BINARYOP(crcc_w_w_w, CRCC_W_W_W)
5810#undef CRC_CASE_EXT_BINARYOP
5811
5812#define CRC_CASE_EXT_UNARYOP(NAME, NODE) \
5813 case Intrinsic::loongarch_##NAME: { \
5814 SDValue NODE = DAG.getNode( \
5815 LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5816 {Chain, Op2, \
5817 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3))}); \
5818 Results.push_back(DAG.getNode(ISD::TRUNCATE, DL, VT, NODE.getValue(0))); \
5819 Results.push_back(NODE.getValue(1)); \
5820 break; \
5821 }
5822 CRC_CASE_EXT_UNARYOP(crc_w_d_w, CRC_W_D_W)
5823 CRC_CASE_EXT_UNARYOP(crcc_w_d_w, CRCC_W_D_W)
5824#undef CRC_CASE_EXT_UNARYOP
5825#define CSR_CASE(ID) \
5826 case Intrinsic::loongarch_##ID: { \
5827 if (!Subtarget.is64Bit()) \
5828 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqLA64); \
5829 break; \
5830 }
5831 CSR_CASE(csrrd_d);
5832 CSR_CASE(csrwr_d);
5833 CSR_CASE(csrxchg_d);
5834 CSR_CASE(iocsrrd_d);
5835#undef CSR_CASE
5836 case Intrinsic::loongarch_csrrd_w: {
5837 unsigned Imm = Op2->getAsZExtVal();
5838 if (!isUInt<14>(Imm)) {
5839 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5840 return;
5841 }
5842 SDValue CSRRDResults =
5843 DAG.getNode(LoongArchISD::CSRRD, DL, {GRLenVT, MVT::Other},
5844 {Chain, DAG.getConstant(Imm, DL, GRLenVT)});
5845 Results.push_back(
5846 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRRDResults.getValue(0)));
5847 Results.push_back(CSRRDResults.getValue(1));
5848 break;
5849 }
5850 case Intrinsic::loongarch_csrwr_w: {
5851 unsigned Imm = N->getConstantOperandVal(3);
5852 if (!isUInt<14>(Imm)) {
5853 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5854 return;
5855 }
5856 SDValue CSRWRResults =
5857 DAG.getNode(LoongArchISD::CSRWR, DL, {GRLenVT, MVT::Other},
5858 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2),
5859 DAG.getConstant(Imm, DL, GRLenVT)});
5860 Results.push_back(
5861 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRWRResults.getValue(0)));
5862 Results.push_back(CSRWRResults.getValue(1));
5863 break;
5864 }
5865 case Intrinsic::loongarch_csrxchg_w: {
5866 unsigned Imm = N->getConstantOperandVal(4);
5867 if (!isUInt<14>(Imm)) {
5868 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgOOR);
5869 return;
5870 }
5871 SDValue CSRXCHGResults = DAG.getNode(
5872 LoongArchISD::CSRXCHG, DL, {GRLenVT, MVT::Other},
5873 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2),
5874 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, N->getOperand(3)),
5875 DAG.getConstant(Imm, DL, GRLenVT)});
5876 Results.push_back(
5877 DAG.getNode(ISD::TRUNCATE, DL, VT, CSRXCHGResults.getValue(0)));
5878 Results.push_back(CSRXCHGResults.getValue(1));
5879 break;
5880 }
5881#define IOCSRRD_CASE(NAME, NODE) \
5882 case Intrinsic::loongarch_##NAME: { \
5883 SDValue IOCSRRDResults = \
5884 DAG.getNode(LoongArchISD::NODE, DL, {MVT::i64, MVT::Other}, \
5885 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2)}); \
5886 Results.push_back( \
5887 DAG.getNode(ISD::TRUNCATE, DL, VT, IOCSRRDResults.getValue(0))); \
5888 Results.push_back(IOCSRRDResults.getValue(1)); \
5889 break; \
5890 }
5891 IOCSRRD_CASE(iocsrrd_b, IOCSRRD_B);
5892 IOCSRRD_CASE(iocsrrd_h, IOCSRRD_H);
5893 IOCSRRD_CASE(iocsrrd_w, IOCSRRD_W);
5894#undef IOCSRRD_CASE
5895 case Intrinsic::loongarch_cpucfg: {
5896 SDValue CPUCFGResults =
5897 DAG.getNode(LoongArchISD::CPUCFG, DL, {GRLenVT, MVT::Other},
5898 {Chain, DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op2)});
5899 Results.push_back(
5900 DAG.getNode(ISD::TRUNCATE, DL, VT, CPUCFGResults.getValue(0)));
5901 Results.push_back(CPUCFGResults.getValue(1));
5902 break;
5903 }
5904 case Intrinsic::loongarch_lddir_d: {
5905 if (!Subtarget.is64Bit()) {
5906 emitErrorAndReplaceIntrinsicResults(N, Results, DAG, ErrorMsgReqLA64);
5907 return;
5908 }
5909 break;
5910 }
5911 }
5912 break;
5913 }
5914 case ISD::READ_REGISTER: {
5915 if (Subtarget.is64Bit())
5916 DAG.getContext()->emitError(
5917 "On LA64, only 64-bit registers can be read.");
5918 else
5919 DAG.getContext()->emitError(
5920 "On LA32, only 32-bit registers can be read.");
5921 Results.push_back(DAG.getUNDEF(VT));
5922 Results.push_back(N->getOperand(0));
5923 break;
5924 }
5926 replaceINTRINSIC_WO_CHAINResults(N, Results, DAG, Subtarget);
5927 break;
5928 }
5929 case ISD::LROUND: {
5930 SDValue Op0 = N->getOperand(0);
5931 EVT OpVT = Op0.getValueType();
5932 RTLIB::Libcall LC =
5933 OpVT == MVT::f64 ? RTLIB::LROUND_F64 : RTLIB::LROUND_F32;
5934 MakeLibCallOptions CallOptions;
5935 CallOptions.setTypeListBeforeSoften(OpVT, MVT::i64);
5936 SDValue Result = makeLibCall(DAG, LC, MVT::i64, Op0, CallOptions, DL).first;
5937 Result = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Result);
5938 Results.push_back(Result);
5939 break;
5940 }
5941 case ISD::ATOMIC_CMP_SWAP: {
5943 break;
5944 }
5945 case ISD::TRUNCATE: {
5946 MVT VT = N->getSimpleValueType(0);
5947 if (getTypeAction(*DAG.getContext(), VT) != TypeWidenVector)
5948 return;
5949
5950 MVT WidenVT = getTypeToTransformTo(*DAG.getContext(), VT).getSimpleVT();
5951 SDValue In = N->getOperand(0);
5952 EVT InVT = In.getValueType();
5953 EVT InEltVT = InVT.getVectorElementType();
5954 EVT EltVT = VT.getVectorElementType();
5955 unsigned MinElts = VT.getVectorNumElements();
5956 unsigned WidenNumElts = WidenVT.getVectorNumElements();
5957 unsigned InBits = InVT.getSizeInBits();
5958
5959 // v8i64 -> (v8i32) -> v8i8
5960 if (InVT == MVT::v8i64 && WidenVT.is128BitVector()) {
5961 InVT = MVT::getVectorVT(MVT::getIntegerVT(256 / MinElts), MinElts);
5962 In = DAG.getNode(N->getOpcode(), DL, InVT, In);
5963 InBits = 256;
5964 }
5965
5966 // v8i32 -> v8i8 / v4i64 -> v4i16 / v4i64 -> v4i8
5967 if ((InVT == MVT::v8i32 || InVT == MVT::v4i64) &&
5968 WidenVT.is128BitVector()) {
5969 InVT = MVT::getVectorVT(MVT::getIntegerVT(128 / MinElts), MinElts);
5970 In = DAG.getNode(N->getOpcode(), DL, InVT, In);
5971 InBits = 128;
5972 InEltVT = InVT.getVectorElementType();
5973 }
5974
5975 if ((128 % InBits) == 0 && WidenVT.is128BitVector()) {
5976 if ((InEltVT.getSizeInBits() % EltVT.getSizeInBits()) == 0) {
5977 int Scale = InEltVT.getSizeInBits() / EltVT.getSizeInBits();
5978 SmallVector<int, 16> TruncMask(WidenNumElts, -1);
5979 for (unsigned I = 0; I < MinElts; ++I)
5980 TruncMask[I] = Scale * I;
5981
5982 unsigned WidenNumElts = 128 / In.getScalarValueSizeInBits();
5983 MVT SVT = In.getSimpleValueType().getScalarType();
5984 MVT VT = MVT::getVectorVT(SVT, WidenNumElts);
5985 SDValue WidenIn =
5986 DAG.getNode(ISD::INSERT_SUBVECTOR, DL, VT, DAG.getUNDEF(VT), In,
5987 DAG.getVectorIdxConstant(0, DL));
5988 assert(isTypeLegal(WidenVT) && isTypeLegal(WidenIn.getValueType()) &&
5989 "Illegal vector type in truncation");
5990 WidenIn = DAG.getBitcast(WidenVT, WidenIn);
5991 Results.push_back(
5992 DAG.getVectorShuffle(WidenVT, DL, WidenIn, WidenIn, TruncMask));
5993 return;
5994 }
5995 }
5996
5997 break;
5998 }
5999 case ISD::SIGN_EXTEND: {
6000 // LASX has native VEXT2XV_* for sign extension.
6001 if (!Subtarget.hasExtLSX() || Subtarget.hasExtLASX())
6002 return;
6003
6004 EVT DstVT = N->getValueType(0);
6005 SDValue Src = N->getOperand(0);
6006 MVT SrcVT = Src.getSimpleValueType();
6007
6008 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
6009 unsigned DstEltBits = DstVT.getScalarSizeInBits();
6010 unsigned NumElts = DstVT.getVectorNumElements();
6011
6012 if (SrcVT.getSizeInBits() > 128)
6013 return;
6014
6015 if (!DstVT.isVector() || DstVT.getSizeInBits() <= 128)
6016 return;
6017
6018 // Legalize and extend the src to 128-bit first.
6019 if (SrcVT.getSizeInBits() < 128) {
6020 unsigned WidenSrcElts = 128 / SrcEltBits;
6021 MVT WidenSrcVT = MVT::getVectorVT(SrcVT.getScalarType(), WidenSrcElts);
6022 Src = DAG.getNode(ISD::INSERT_SUBVECTOR, DL, WidenSrcVT,
6023 DAG.getUNDEF(WidenSrcVT), Src,
6024 DAG.getVectorIdxConstant(0, DL));
6025 SrcVT = WidenSrcVT;
6026
6027 unsigned FirstStageEltBits = 128 / NumElts;
6028 MVT FirstStageEltVT = MVT::getIntegerVT(FirstStageEltBits);
6029 MVT FirstStageVT = MVT::getVectorVT(FirstStageEltVT, NumElts);
6030 Src = DAG.getNode(ISD::SIGN_EXTEND_VECTOR_INREG, DL, FirstStageVT, Src);
6031 SrcVT = FirstStageVT;
6032 SrcEltBits = FirstStageEltBits;
6033 }
6034
6036 Blocks.push_back(Src);
6037
6038 // Sign-extend the src by using SLTI + VILVL + VILVH recursively.
6039 while (SrcEltBits < DstEltBits) {
6040 unsigned NextEltBits = SrcEltBits * 2;
6041 MVT NextEltVT = MVT::getIntegerVT(NextEltBits);
6042 unsigned CurEltsPerBlock = SrcVT.getVectorNumElements();
6043 unsigned NextEltsPerBlock = CurEltsPerBlock / 2;
6044 MVT NextBlockVT = MVT::getVectorVT(NextEltVT, NextEltsPerBlock);
6045
6046 SmallVector<SDValue, 8> NextBlocks;
6047 NextBlocks.reserve(Blocks.size() * 2);
6048 for (SDValue Block : Blocks) {
6049 SDValue Zero = DAG.getConstant(0, DL, SrcVT);
6050 SDValue Mask = DAG.getNode(ISD::SETCC, DL, SrcVT, Block, Zero,
6051 DAG.getCondCode(ISD::SETLT));
6052 SDValue LoInterleaved =
6053 DAG.getNode(LoongArchISD::VILVL, DL, SrcVT, Mask, Block);
6054 SDValue HiInterleaved =
6055 DAG.getNode(LoongArchISD::VILVH, DL, SrcVT, Mask, Block);
6056
6057 NextBlocks.push_back(DAG.getBitcast(NextBlockVT, LoInterleaved));
6058 NextBlocks.push_back(DAG.getBitcast(NextBlockVT, HiInterleaved));
6059 }
6060
6061 Blocks = std::move(NextBlocks);
6062 SrcVT = NextBlockVT;
6063 SrcEltBits = NextEltBits;
6064 }
6065
6066 Results.push_back(DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Blocks));
6067 break;
6068 }
6069 case ISD::FP_EXTEND:
6070 // FP_EXTEND may reach here due to the Custom action for v2f32 results, but
6071 // no target-specific lowering is required. Leave it unchanged and rely on
6072 // the default type legalization.
6073 break;
6074 }
6075}
6076
6077/// Try to fold: (and (xor X, -1), Y) -> (vandn X, Y).
6079 SelectionDAG &DAG) {
6080 assert(N->getOpcode() == ISD::AND && "Unexpected opcode combine into ANDN");
6081
6082 MVT VT = N->getSimpleValueType(0);
6083 if (!VT.is128BitVector() && !VT.is256BitVector())
6084 return SDValue();
6085
6086 SDValue X, Y;
6087 SDValue N0 = N->getOperand(0);
6088 SDValue N1 = N->getOperand(1);
6089
6090 if (SDValue Not = isNOT(N0, DAG)) {
6091 X = Not;
6092 Y = N1;
6093 } else if (SDValue Not = isNOT(N1, DAG)) {
6094 X = Not;
6095 Y = N0;
6096 } else
6097 return SDValue();
6098
6099 X = DAG.getBitcast(VT, X);
6100 Y = DAG.getBitcast(VT, Y);
6101 return DAG.getNode(LoongArchISD::VANDN, DL, VT, X, Y);
6102}
6103
6104static bool isConstantSplatVector(SDValue N, APInt &SplatValue,
6105 unsigned MinSizeInBits) {
6108
6109 if (!Node)
6110 return false;
6111
6112 APInt SplatUndef;
6113 unsigned SplatBitSize;
6114 bool HasAnyUndefs;
6115
6116 return Node->isConstantSplat(SplatValue, SplatUndef, SplatBitSize,
6117 HasAnyUndefs, MinSizeInBits,
6118 /*IsBigEndian=*/false);
6119}
6120
6121static SDValue matchDeinterleaveBuildVector(SDValue N, unsigned &StartIndex) {
6122 auto *BV = dyn_cast<BuildVectorSDNode>(N);
6123 if (!BV)
6124 return SDValue();
6125
6126 SDValue Src;
6127 int Start = -1;
6128
6129 for (unsigned i = 0, NumElts = BV->getNumOperands(); i < NumElts; ++i) {
6130 SDValue Op = BV->getOperand(i);
6131 if (Op.isUndef())
6132 continue;
6133 if (Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT)
6134 return SDValue();
6135
6136 auto *IdxC = dyn_cast<ConstantSDNode>(Op.getOperand(1));
6137 if (!IdxC)
6138 return SDValue();
6139
6140 unsigned EltIdx = IdxC->getZExtValue();
6141 if (Start < 0)
6142 Start = (int)EltIdx - (int)(i * 2);
6143 if (Start < 0 || Start > 1 || EltIdx != (unsigned)(Start + (int)(i * 2)))
6144 return SDValue();
6145
6146 SDValue CurSrc = Op.getOperand(0);
6147 if (!Src)
6148 Src = CurSrc;
6149 else if (Src != CurSrc)
6150 return SDValue();
6151 }
6152
6153 if (!Src || Start < 0)
6154 return SDValue();
6155
6156 StartIndex = (unsigned)Start;
6157 return Src;
6158}
6159
6160static SDValue
6162 const LoongArchSubtarget &Subtarget) {
6163 if (!Subtarget.hasExtLSX())
6164 return SDValue();
6165
6166 unsigned Opc = N->getOpcode();
6167 assert((Opc == ISD::ADD || Opc == ISD::SUB) && "Unexpected opcode");
6168
6169 EVT VT = N->getValueType(0);
6170 SDLoc DL(N);
6171
6172 SDValue LHS = N->getOperand(0);
6173 SDValue RHS = N->getOperand(1);
6174
6175 bool isSigned;
6176 unsigned ExtOpc = LHS.getOpcode();
6177 if (ExtOpc == ISD::SIGN_EXTEND)
6178 isSigned = true;
6179 else if (ExtOpc == ISD::ZERO_EXTEND)
6180 isSigned = false;
6181 else
6182 return SDValue();
6183
6184 if (ExtOpc != RHS.getOpcode())
6185 return SDValue();
6186
6187 if (!LHS.hasOneUse() || !RHS.hasOneUse())
6188 return SDValue();
6189
6190 unsigned OddIdx, EvenIdx;
6191 SDValue LHSVec = matchDeinterleaveBuildVector(LHS.getOperand(0), OddIdx);
6192 SDValue RHSVec = matchDeinterleaveBuildVector(RHS.getOperand(0), EvenIdx);
6193
6194 if (!LHSVec || !RHSVec)
6195 return SDValue();
6196 if (OddIdx != 1 || EvenIdx != 0)
6197 return SDValue();
6198 if (LHSVec.getValueType() != RHSVec.getValueType())
6199 return SDValue();
6200
6201 EVT SrcVT = LHSVec.getValueType();
6202 EVT SrcEltVT = SrcVT.getVectorElementType();
6203 EVT DstEltVT = VT.getVectorElementType();
6204 auto &TLI = DAG.getTargetLoweringInfo();
6205
6206 if (!TLI.isTypeLegal(VT) || !TLI.isTypeLegal(SrcVT))
6207 return SDValue();
6208 if (!SrcVT.isVector() || !VT.isVector())
6209 return SDValue();
6210 if (SrcVT.getSizeInBits() != VT.getSizeInBits())
6211 return SDValue();
6212 if (DstEltVT.getSizeInBits() != SrcEltVT.getSizeInBits() * 2)
6213 return SDValue();
6214 if (!SrcEltVT.isInteger() || SrcEltVT.getSizeInBits() > 32)
6215 return SDValue();
6216
6217 unsigned TargetOpc;
6218 if (Opc == ISD::ADD)
6219 TargetOpc = isSigned ? LoongArchISD::VHADDW : LoongArchISD::VHADDW_U;
6220 else
6221 TargetOpc = isSigned ? LoongArchISD::VHSUBW : LoongArchISD::VHSUBW_U;
6222
6223 return DAG.getNode(TargetOpc, DL, VT, LHSVec, RHSVec);
6224}
6225
6228 const LoongArchSubtarget &Subtarget) {
6229 if (SDValue V = performHorizWideningCombine(N, DAG, Subtarget))
6230 return V;
6231
6232 if (DCI.isBeforeLegalizeOps())
6233 return SDValue();
6234
6235 EVT VT = N->getValueType(0);
6236 if (!VT.isVector())
6237 return SDValue();
6238
6239 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
6240 return SDValue();
6241
6242 EVT EltVT = VT.getVectorElementType();
6243 if (!EltVT.isInteger())
6244 return SDValue();
6245
6246 // match:
6247 //
6248 // add
6249 // (and
6250 // (srl X, shift-1) / X
6251 // 1)
6252 // (srl/sra X, shift)
6253
6254 SDValue Add0 = N->getOperand(0);
6255 SDValue Add1 = N->getOperand(1);
6256 SDValue And;
6257 SDValue Shr;
6258
6259 if (Add0.getOpcode() == ISD::AND) {
6260 And = Add0;
6261 Shr = Add1;
6262 } else if (Add1.getOpcode() == ISD::AND) {
6263 And = Add1;
6264 Shr = Add0;
6265 } else {
6266 return SDValue();
6267 }
6268
6269 // match:
6270 //
6271 // srl/sra X, shift
6272
6273 if (Shr.getOpcode() != ISD::SRL && Shr.getOpcode() != ISD::SRA)
6274 return SDValue();
6275
6276 SDValue X = Shr.getOperand(0);
6277 SDValue Shift = Shr.getOperand(1);
6278 APInt ShiftVal;
6279
6280 if (!isConstantSplatVector(Shift, ShiftVal, EltVT.getSizeInBits()))
6281 return SDValue();
6282
6283 if (ShiftVal == 0)
6284 return SDValue();
6285
6286 // match:
6287 //
6288 // and
6289 // (srl X, shift-1) / X
6290 // 1
6291
6292 SDValue One = And.getOperand(1);
6293 APInt SplatVal;
6294
6295 if (!isConstantSplatVector(One, SplatVal, EltVT.getSizeInBits()))
6296 return SDValue();
6297
6298 if (SplatVal != 1)
6299 return SDValue();
6300
6301 if (And.getOperand(0) == X) {
6302 // match:
6303 //
6304 // shift == 1
6305
6306 if (ShiftVal != 1)
6307 return SDValue();
6308 } else {
6309 // match:
6310 //
6311 // srl X, shift-1
6312
6313 SDValue Srl = And.getOperand(0);
6314
6315 if (Srl.getOpcode() != ISD::SRL)
6316 return SDValue();
6317
6318 if (Srl.getOperand(0) != X)
6319 return SDValue();
6320
6321 // match:
6322 //
6323 // shift-1
6324
6325 SDValue ShiftMinus1 = Srl.getOperand(1);
6326
6327 if (!isConstantSplatVector(ShiftMinus1, SplatVal, EltVT.getSizeInBits()))
6328 return SDValue();
6329
6330 if (ShiftVal != (SplatVal + 1))
6331 return SDValue();
6332 }
6333
6334 // We matched a rounded right shift pattern and can lower it
6335 // to a single vector rounded shift instruction.
6336
6337 SDLoc DL(N);
6338 return DAG.getNode(Shr.getOpcode() == ISD::SRL ? LoongArchISD::VSRLR
6339 : LoongArchISD::VSRAR,
6340 DL, VT, X, Shift);
6341}
6342
6345 const LoongArchSubtarget &Subtarget) {
6346 if (DCI.isBeforeLegalizeOps())
6347 return SDValue();
6348
6349 SDValue FirstOperand = N->getOperand(0);
6350 SDValue SecondOperand = N->getOperand(1);
6351 unsigned FirstOperandOpc = FirstOperand.getOpcode();
6352 EVT ValTy = N->getValueType(0);
6353 SDLoc DL(N);
6354 uint64_t lsb, msb;
6355 unsigned SMIdx, SMLen;
6356 ConstantSDNode *CN;
6357 SDValue NewOperand;
6358 MVT GRLenVT = Subtarget.getGRLenVT();
6359
6360 if (SDValue R = combineAndNotIntoVANDN(N, DL, DAG))
6361 return R;
6362
6363 // BSTRPICK requires the 32S feature.
6364 if (!Subtarget.has32S())
6365 return SDValue();
6366
6367 // Op's second operand must be a shifted mask.
6368 if (!(CN = dyn_cast<ConstantSDNode>(SecondOperand)) ||
6369 !isShiftedMask_64(CN->getZExtValue(), SMIdx, SMLen))
6370 return SDValue();
6371
6372 if (FirstOperandOpc == ISD::SRA || FirstOperandOpc == ISD::SRL) {
6373 // Pattern match BSTRPICK.
6374 // $dst = and ((sra or srl) $src , lsb), (2**len - 1)
6375 // => BSTRPICK $dst, $src, msb, lsb
6376 // where msb = lsb + len - 1
6377
6378 // The second operand of the shift must be an immediate.
6379 if (!(CN = dyn_cast<ConstantSDNode>(FirstOperand.getOperand(1))))
6380 return SDValue();
6381
6382 lsb = CN->getZExtValue();
6383
6384 // Return if the shifted mask does not start at bit 0 or the sum of its
6385 // length and lsb exceeds the word's size.
6386 if (SMIdx != 0 || lsb + SMLen > ValTy.getSizeInBits())
6387 return SDValue();
6388
6389 NewOperand = FirstOperand.getOperand(0);
6390 } else {
6391 // Pattern match BSTRPICK.
6392 // $dst = and $src, (2**len- 1) , if len > 12
6393 // => BSTRPICK $dst, $src, msb, lsb
6394 // where lsb = 0 and msb = len - 1
6395
6396 // If the mask is <= 0xfff, andi can be used instead.
6397 if (CN->getZExtValue() <= 0xfff)
6398 return SDValue();
6399
6400 // Return if the MSB exceeds.
6401 if (SMIdx + SMLen > ValTy.getSizeInBits())
6402 return SDValue();
6403
6404 if (SMIdx > 0) {
6405 // Omit if the constant has more than 2 uses. This a conservative
6406 // decision. Whether it is a win depends on the HW microarchitecture.
6407 // However it should always be better for 1 and 2 uses.
6408 if (CN->use_size() > 2)
6409 return SDValue();
6410 // Return if the constant can be composed by a single LU12I.W.
6411 if ((CN->getZExtValue() & 0xfff) == 0)
6412 return SDValue();
6413 // Return if the constand can be composed by a single ADDI with
6414 // the zero register.
6415 if (CN->getSExtValue() >= -2048 && CN->getSExtValue() < 0)
6416 return SDValue();
6417 }
6418
6419 lsb = SMIdx;
6420 NewOperand = FirstOperand;
6421 }
6422
6423 msb = lsb + SMLen - 1;
6424 SDValue NR0 = DAG.getNode(LoongArchISD::BSTRPICK, DL, ValTy, NewOperand,
6425 DAG.getConstant(msb, DL, GRLenVT),
6426 DAG.getConstant(lsb, DL, GRLenVT));
6427 if (FirstOperandOpc == ISD::SRA || FirstOperandOpc == ISD::SRL || lsb == 0)
6428 return NR0;
6429 // Try to optimize to
6430 // bstrpick $Rd, $Rs, msb, lsb
6431 // slli $Rd, $Rd, lsb
6432 return DAG.getNode(ISD::SHL, DL, ValTy, NR0,
6433 DAG.getConstant(lsb, DL, GRLenVT));
6434}
6435
6436// Return the original source vector if N consists of the half
6437// of each 128-bit lane.
6440
6441 EVT DstVT = N.getValueType();
6442 if (!DstVT.isVector())
6443 return SDValue();
6444
6445 unsigned NumElts = DstVT.getVectorNumElements();
6446
6447 // LSX canonical form:
6448 if (N.getOpcode() == ISD::EXTRACT_SUBVECTOR) {
6449 SDValue Src = N.getOperand(0);
6450 EVT SrcVT = Src.getValueType();
6451
6452 if (!SrcVT.isVector() || !SrcVT.is128BitVector())
6453 return SDValue();
6454 if (SrcVT.getSizeInBits() != DstVT.getSizeInBits() * 2)
6455 return SDValue();
6456 if (SrcVT.getVectorNumElements() != NumElts * 2)
6457 return SDValue();
6458 if (N.getConstantOperandVal(1) != (isLow ? 0 : NumElts))
6459 return SDValue();
6460
6461 return Src;
6462 }
6463
6464 // LASX canonical form:
6465 auto *BV = dyn_cast<BuildVectorSDNode>(N);
6466 if (!BV)
6467 return SDValue();
6468
6469 if (NumElts % 2 != 0)
6470 return SDValue();
6471
6472 SDValue Src;
6473 EVT SrcVT;
6474
6475 for (unsigned I = 0; I != NumElts; ++I) {
6476 SDValue Elt = BV->getOperand(I);
6477 if (Elt.isUndef())
6478 continue;
6480 return SDValue();
6481
6482 SDValue ThisSrc = Elt.getOperand(0);
6483 SDValue Idx = Elt.getOperand(1);
6484 auto *CI = dyn_cast<ConstantSDNode>(Idx);
6485 if (!CI)
6486 return SDValue();
6487
6488 if (!Src) {
6489 Src = ThisSrc;
6490 SrcVT = Src.getValueType();
6491 if (!SrcVT.isVector())
6492 return SDValue();
6493
6494 if (!SrcVT.is256BitVector())
6495 return SDValue();
6496 if (SrcVT.getSizeInBits() != DstVT.getSizeInBits() * 2)
6497 return SDValue();
6498 if (SrcVT.getVectorNumElements() != NumElts * 2)
6499 return SDValue();
6500 } else if (ThisSrc != Src) {
6501 return SDValue();
6502 }
6503
6504 unsigned Half = NumElts / 2;
6505 unsigned ExpectedIdx = (I < Half) ? I : (I + Half);
6506 ExpectedIdx += isLow ? 0 : Half;
6507
6508 if (CI->getZExtValue() != ExpectedIdx)
6509 return SDValue();
6510 }
6511
6512 return Src;
6513}
6514
6517 const LoongArchSubtarget &Subtarget) {
6518 assert(N->getOpcode() == ISD::SHL && "Unexpected opcode");
6519
6520 EVT VT = N->getValueType(0);
6521 SDLoc DL(N);
6522
6523 SDValue LHS = N->getOperand(0);
6524 SDValue RHS = N->getOperand(1);
6525
6526 bool isSigned;
6527 unsigned ExtOpc = LHS.getOpcode();
6528 if (ExtOpc == ISD::SIGN_EXTEND)
6529 isSigned = true;
6530 else if (ExtOpc == ISD::ZERO_EXTEND)
6531 isSigned = false;
6532 else
6533 return SDValue();
6534
6535 if (!LHS.hasOneUse())
6536 return SDValue();
6537
6538 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT) ||
6539 N->getValueSizeInBits(0) != LHS->getOperand(0).getValueSizeInBits() * 2)
6540 return SDValue();
6541
6542 SDValue Vec = matchHalfOf128BitLanes(LHS.getOperand(0), /*isLow=*/true);
6543 if (!Vec)
6544 return SDValue();
6545
6546 EVT SrcVT = Vec.getValueType();
6547 EVT SrcEltVT = SrcVT.getVectorElementType();
6548 EVT DstEltVT = VT.getVectorElementType();
6549 APInt Imm;
6550 if (!isConstantSplatVector(RHS, Imm, DstEltVT.getSizeInBits()))
6551 return SDValue();
6552 if (!Imm.ult(SrcEltVT.getSizeInBits()))
6553 return SDValue();
6554
6555 unsigned Opc = isSigned ? LoongArchISD::VSLLWIL : LoongArchISD::VSLLWIL_U;
6556 SDValue Sht = DAG.getConstant(Imm.getZExtValue(), DL, Subtarget.getGRLenVT());
6557 return DAG.getNode(Opc, DL, VT, Vec, Sht);
6558}
6559
6562 const LoongArchSubtarget &Subtarget) {
6563 // BSTRPICK requires the 32S feature.
6564 if (!Subtarget.has32S())
6565 return SDValue();
6566
6567 if (DCI.isBeforeLegalizeOps())
6568 return SDValue();
6569
6570 // $dst = srl (and $src, Mask), Shamt
6571 // =>
6572 // BSTRPICK $dst, $src, MaskIdx+MaskLen-1, Shamt
6573 // when Mask is a shifted mask, and MaskIdx <= Shamt <= MaskIdx+MaskLen-1
6574 //
6575
6576 SDValue FirstOperand = N->getOperand(0);
6577 ConstantSDNode *CN;
6578 EVT ValTy = N->getValueType(0);
6579 SDLoc DL(N);
6580 MVT GRLenVT = Subtarget.getGRLenVT();
6581 unsigned MaskIdx, MaskLen;
6582 uint64_t Shamt;
6583
6584 // The first operand must be an AND and the second operand of the AND must be
6585 // a shifted mask.
6586 if (FirstOperand.getOpcode() != ISD::AND ||
6587 !(CN = dyn_cast<ConstantSDNode>(FirstOperand.getOperand(1))) ||
6588 !isShiftedMask_64(CN->getZExtValue(), MaskIdx, MaskLen))
6589 return SDValue();
6590
6591 // The second operand (shift amount) must be an immediate.
6592 if (!(CN = dyn_cast<ConstantSDNode>(N->getOperand(1))))
6593 return SDValue();
6594
6595 Shamt = CN->getZExtValue();
6596 if (MaskIdx <= Shamt && Shamt <= MaskIdx + MaskLen - 1)
6597 return DAG.getNode(LoongArchISD::BSTRPICK, DL, ValTy,
6598 FirstOperand->getOperand(0),
6599 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
6600 DAG.getConstant(Shamt, DL, GRLenVT));
6601
6602 return SDValue();
6603}
6604
6607 const LoongArchSubtarget &Subtarget) {
6608 if (SDValue V = performHorizWideningCombine(N, DAG, Subtarget))
6609 return V;
6610
6611 return SDValue();
6612}
6613
6614// Helper to peek through bitops/trunc/setcc to determine size of source vector.
6615// Allows BITCASTCombine to determine what size vector generated a <X x i1>.
6616static bool checkBitcastSrcVectorSize(SDValue Src, unsigned Size,
6617 unsigned Depth) {
6618 // Limit recursion.
6620 return false;
6621 switch (Src.getOpcode()) {
6622 case ISD::SETCC:
6623 case ISD::TRUNCATE:
6624 return Src.getOperand(0).getValueSizeInBits() == Size;
6625 case ISD::FREEZE:
6626 return checkBitcastSrcVectorSize(Src.getOperand(0), Size, Depth + 1);
6627 case ISD::AND:
6628 case ISD::XOR:
6629 case ISD::OR:
6630 return checkBitcastSrcVectorSize(Src.getOperand(0), Size, Depth + 1) &&
6631 checkBitcastSrcVectorSize(Src.getOperand(1), Size, Depth + 1);
6632 case ISD::SELECT:
6633 case ISD::VSELECT:
6634 return Src.getOperand(0).getScalarValueSizeInBits() == 1 &&
6635 checkBitcastSrcVectorSize(Src.getOperand(1), Size, Depth + 1) &&
6636 checkBitcastSrcVectorSize(Src.getOperand(2), Size, Depth + 1);
6637 case ISD::BUILD_VECTOR:
6638 return ISD::isBuildVectorAllZeros(Src.getNode()) ||
6639 ISD::isBuildVectorAllOnes(Src.getNode());
6640 }
6641 return false;
6642}
6643
6644// Helper to push sign extension of vXi1 SETCC result through bitops.
6646 SDValue Src, const SDLoc &DL) {
6647 switch (Src.getOpcode()) {
6648 case ISD::SETCC:
6649 case ISD::FREEZE:
6650 case ISD::TRUNCATE:
6651 case ISD::BUILD_VECTOR:
6652 return DAG.getNode(ISD::SIGN_EXTEND, DL, SExtVT, Src);
6653 case ISD::AND:
6654 case ISD::XOR:
6655 case ISD::OR:
6656 return DAG.getNode(
6657 Src.getOpcode(), DL, SExtVT,
6658 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(0), DL),
6659 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(1), DL));
6660 case ISD::SELECT:
6661 case ISD::VSELECT:
6662 return DAG.getSelect(
6663 DL, SExtVT, Src.getOperand(0),
6664 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(1), DL),
6665 signExtendBitcastSrcVector(DAG, SExtVT, Src.getOperand(2), DL));
6666 }
6667 llvm_unreachable("Unexpected node type for vXi1 sign extension");
6668}
6669
6670static SDValue
6673 const LoongArchSubtarget &Subtarget) {
6674 SDLoc DL(N);
6675 EVT VT = N->getValueType(0);
6676 SDValue Src = N->getOperand(0);
6677 EVT SrcVT = Src.getValueType();
6678
6679 if (Src.getOpcode() != ISD::SETCC || !Src.hasOneUse())
6680 return SDValue();
6681
6682 bool UseLASX;
6683 unsigned Opc = ISD::DELETED_NODE;
6684 EVT CmpVT = Src.getOperand(0).getValueType();
6685 EVT EltVT = CmpVT.getVectorElementType();
6686
6687 if (Subtarget.hasExtLSX() && CmpVT.getSizeInBits() == 128)
6688 UseLASX = false;
6689 else if (Subtarget.has32S() && Subtarget.hasExtLASX() &&
6690 CmpVT.getSizeInBits() == 256)
6691 UseLASX = true;
6692 else
6693 return SDValue();
6694
6695 SDValue SrcN1 = Src.getOperand(1);
6696 switch (cast<CondCodeSDNode>(Src.getOperand(2))->get()) {
6697 default:
6698 break;
6699 case ISD::SETEQ:
6700 // x == 0 => not (vmsknez.b x)
6701 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6702 Opc = UseLASX ? LoongArchISD::XVMSKEQZ : LoongArchISD::VMSKEQZ;
6703 break;
6704 case ISD::SETGT:
6705 // x > -1 => vmskgez.b x
6706 if (ISD::isBuildVectorAllOnes(SrcN1.getNode()) && EltVT == MVT::i8)
6707 Opc = UseLASX ? LoongArchISD::XVMSKGEZ : LoongArchISD::VMSKGEZ;
6708 break;
6709 case ISD::SETGE:
6710 // x >= 0 => vmskgez.b x
6711 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6712 Opc = UseLASX ? LoongArchISD::XVMSKGEZ : LoongArchISD::VMSKGEZ;
6713 break;
6714 case ISD::SETLT:
6715 // x < 0 => vmskltz.{b,h,w,d} x
6716 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) &&
6717 (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
6718 EltVT == MVT::i64))
6719 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6720 break;
6721 case ISD::SETLE:
6722 // x <= -1 => vmskltz.{b,h,w,d} x
6723 if (ISD::isBuildVectorAllOnes(SrcN1.getNode()) &&
6724 (EltVT == MVT::i8 || EltVT == MVT::i16 || EltVT == MVT::i32 ||
6725 EltVT == MVT::i64))
6726 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6727 break;
6728 case ISD::SETNE:
6729 // x != 0 => vmsknez.b x
6730 if (ISD::isBuildVectorAllZeros(SrcN1.getNode()) && EltVT == MVT::i8)
6731 Opc = UseLASX ? LoongArchISD::XVMSKNEZ : LoongArchISD::VMSKNEZ;
6732 break;
6733 }
6734
6735 if (Opc == ISD::DELETED_NODE)
6736 return SDValue();
6737
6738 SDValue V = DAG.getNode(Opc, DL, Subtarget.getGRLenVT(), Src.getOperand(0));
6740 V = DAG.getZExtOrTrunc(V, DL, T);
6741 return DAG.getBitcast(VT, V);
6742}
6743
6746 const LoongArchSubtarget &Subtarget) {
6747 SDLoc DL(N);
6748 EVT VT = N->getValueType(0);
6749 SDValue Src = N->getOperand(0);
6750 EVT SrcVT = Src.getValueType();
6751 MVT GRLenVT = Subtarget.getGRLenVT();
6752
6753 if (!DCI.isBeforeLegalizeOps())
6754 return SDValue();
6755
6756 if (!SrcVT.isSimple() || SrcVT.getScalarType() != MVT::i1)
6757 return SDValue();
6758
6759 // Combine SETCC and BITCAST into [X]VMSK{LT,GE,NE} when possible
6760 SDValue Res = performSETCC_BITCASTCombine(N, DAG, DCI, Subtarget);
6761 if (Res)
6762 return Res;
6763
6764 // Generate vXi1 using [X]VMSKLTZ
6765 MVT SExtVT;
6766 unsigned Opc;
6767 bool UseLASX = false;
6768 bool PropagateSExt = false;
6769
6770 if (Src.getOpcode() == ISD::SETCC && Src.hasOneUse()) {
6771 EVT CmpVT = Src.getOperand(0).getValueType();
6772 if (CmpVT.getSizeInBits() > 256)
6773 return SDValue();
6774 }
6775
6776 switch (SrcVT.getSimpleVT().SimpleTy) {
6777 default:
6778 return SDValue();
6779 case MVT::v2i1:
6780 SExtVT = MVT::v2i64;
6781 break;
6782 case MVT::v4i1:
6783 SExtVT = MVT::v4i32;
6784 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6785 SExtVT = MVT::v4i64;
6786 UseLASX = true;
6787 PropagateSExt = true;
6788 }
6789 break;
6790 case MVT::v8i1:
6791 SExtVT = MVT::v8i16;
6792 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6793 SExtVT = MVT::v8i32;
6794 UseLASX = true;
6795 PropagateSExt = true;
6796 }
6797 break;
6798 case MVT::v16i1:
6799 SExtVT = MVT::v16i8;
6800 if (Subtarget.hasExtLASX() && checkBitcastSrcVectorSize(Src, 256, 0)) {
6801 SExtVT = MVT::v16i16;
6802 UseLASX = true;
6803 PropagateSExt = true;
6804 }
6805 break;
6806 case MVT::v32i1:
6807 SExtVT = MVT::v32i8;
6808 UseLASX = true;
6809 break;
6810 };
6811 Src = PropagateSExt ? signExtendBitcastSrcVector(DAG, SExtVT, Src, DL)
6812 : DAG.getNode(ISD::SIGN_EXTEND, DL, SExtVT, Src);
6813
6814 SDValue V;
6815 if (!Subtarget.has32S() || !Subtarget.hasExtLASX()) {
6816 if (Src.getSimpleValueType() == MVT::v32i8) {
6817 SDValue Lo, Hi;
6818 std::tie(Lo, Hi) = DAG.SplitVector(Src, DL);
6819 Lo = DAG.getNode(LoongArchISD::VMSKLTZ, DL, GRLenVT, Lo);
6820 Hi = DAG.getNode(LoongArchISD::VMSKLTZ, DL, GRLenVT, Hi);
6821 Hi = DAG.getNode(ISD::SHL, DL, GRLenVT, Hi,
6822 DAG.getShiftAmountConstant(16, GRLenVT, DL));
6823 V = DAG.getNode(ISD::OR, DL, GRLenVT, Lo, Hi);
6824 } else if (UseLASX) {
6825 return SDValue();
6826 }
6827 }
6828
6829 if (!V) {
6830 Opc = UseLASX ? LoongArchISD::XVMSKLTZ : LoongArchISD::VMSKLTZ;
6831 V = DAG.getNode(Opc, DL, GRLenVT, Src);
6832 }
6833
6835 V = DAG.getZExtOrTrunc(V, DL, T);
6836 return DAG.getBitcast(VT, V);
6837}
6838
6841 const LoongArchSubtarget &Subtarget) {
6842 MVT GRLenVT = Subtarget.getGRLenVT();
6843 EVT ValTy = N->getValueType(0);
6844 SDValue N0 = N->getOperand(0), N1 = N->getOperand(1);
6845 ConstantSDNode *CN0, *CN1;
6846 SDLoc DL(N);
6847 unsigned ValBits = ValTy.getSizeInBits();
6848 unsigned MaskIdx0, MaskLen0, MaskIdx1, MaskLen1;
6849 unsigned Shamt;
6850 bool SwapAndRetried = false;
6851
6852 // BSTRPICK requires the 32S feature.
6853 if (!Subtarget.has32S())
6854 return SDValue();
6855
6856 if (DCI.isBeforeLegalizeOps())
6857 return SDValue();
6858
6859 if (ValBits != 32 && ValBits != 64)
6860 return SDValue();
6861
6862Retry:
6863 // 1st pattern to match BSTRINS:
6864 // R = or (and X, mask0), (and (shl Y, lsb), mask1)
6865 // where mask1 = (2**size - 1) << lsb, mask0 = ~mask1
6866 // =>
6867 // R = BSTRINS X, Y, msb, lsb (where msb = lsb + size - 1)
6868 if (N0.getOpcode() == ISD::AND &&
6869 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6870 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6871 N1.getOpcode() == ISD::AND && N1.getOperand(0).getOpcode() == ISD::SHL &&
6872 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6873 isShiftedMask_64(CN1->getZExtValue(), MaskIdx1, MaskLen1) &&
6874 MaskIdx0 == MaskIdx1 && MaskLen0 == MaskLen1 &&
6875 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6876 (Shamt = CN1->getZExtValue()) == MaskIdx0 &&
6877 (MaskIdx0 + MaskLen0 <= ValBits)) {
6878 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 1\n");
6879 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6880 N1.getOperand(0).getOperand(0),
6881 DAG.getConstant((MaskIdx0 + MaskLen0 - 1), DL, GRLenVT),
6882 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6883 }
6884
6885 // 2nd pattern to match BSTRINS:
6886 // R = or (and X, mask0), (shl (and Y, mask1), lsb)
6887 // where mask1 = (2**size - 1), mask0 = ~(mask1 << lsb)
6888 // =>
6889 // R = BSTRINS X, Y, msb, lsb (where msb = lsb + size - 1)
6890 if (N0.getOpcode() == ISD::AND &&
6891 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6892 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6893 N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::AND &&
6894 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6895 (Shamt = CN1->getZExtValue()) == MaskIdx0 &&
6896 (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6897 isShiftedMask_64(CN1->getZExtValue(), MaskIdx1, MaskLen1) &&
6898 MaskLen0 == MaskLen1 && MaskIdx1 == 0 &&
6899 (MaskIdx0 + MaskLen0 <= ValBits)) {
6900 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 2\n");
6901 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6902 N1.getOperand(0).getOperand(0),
6903 DAG.getConstant((MaskIdx0 + MaskLen0 - 1), DL, GRLenVT),
6904 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6905 }
6906
6907 // 3rd pattern to match BSTRINS:
6908 // R = or (and X, mask0), (and Y, mask1)
6909 // where ~mask0 = (2**size - 1) << lsb, mask0 & mask1 = 0
6910 // =>
6911 // R = BSTRINS X, (shr (and Y, mask1), lsb), msb, lsb
6912 // where msb = lsb + size - 1
6913 if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::AND &&
6914 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6915 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6916 (MaskIdx0 + MaskLen0 <= 64) &&
6917 (CN1 = dyn_cast<ConstantSDNode>(N1->getOperand(1))) &&
6918 (CN1->getSExtValue() & CN0->getSExtValue()) == 0) {
6919 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 3\n");
6920 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6921 DAG.getNode(ISD::SRL, DL, N1->getValueType(0), N1,
6922 DAG.getConstant(MaskIdx0, DL, GRLenVT)),
6923 DAG.getConstant(ValBits == 32
6924 ? (MaskIdx0 + (MaskLen0 & 31) - 1)
6925 : (MaskIdx0 + MaskLen0 - 1),
6926 DL, GRLenVT),
6927 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6928 }
6929
6930 // 4th pattern to match BSTRINS:
6931 // R = or (and X, mask), (shl Y, shamt)
6932 // where mask = (2**shamt - 1)
6933 // =>
6934 // R = BSTRINS X, Y, ValBits - 1, shamt
6935 // where ValBits = 32 or 64
6936 if (N0.getOpcode() == ISD::AND && N1.getOpcode() == ISD::SHL &&
6937 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6938 isShiftedMask_64(CN0->getZExtValue(), MaskIdx0, MaskLen0) &&
6939 MaskIdx0 == 0 && (CN1 = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6940 (Shamt = CN1->getZExtValue()) == MaskLen0 &&
6941 (MaskIdx0 + MaskLen0 <= ValBits)) {
6942 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 4\n");
6943 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6944 N1.getOperand(0),
6945 DAG.getConstant((ValBits - 1), DL, GRLenVT),
6946 DAG.getConstant(Shamt, DL, GRLenVT));
6947 }
6948
6949 // 5th pattern to match BSTRINS:
6950 // R = or (and X, mask), const
6951 // where ~mask = (2**size - 1) << lsb, mask & const = 0
6952 // =>
6953 // R = BSTRINS X, (const >> lsb), msb, lsb
6954 // where msb = lsb + size - 1
6955 if (N0.getOpcode() == ISD::AND &&
6956 (CN0 = dyn_cast<ConstantSDNode>(N0.getOperand(1))) &&
6957 isShiftedMask_64(~CN0->getSExtValue(), MaskIdx0, MaskLen0) &&
6958 (CN1 = dyn_cast<ConstantSDNode>(N1)) &&
6959 (CN1->getSExtValue() & CN0->getSExtValue()) == 0) {
6960 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 5\n");
6961 return DAG.getNode(
6962 LoongArchISD::BSTRINS, DL, ValTy, N0.getOperand(0),
6963 DAG.getSignedConstant(CN1->getSExtValue() >> MaskIdx0, DL, ValTy),
6964 DAG.getConstant(ValBits == 32 ? (MaskIdx0 + (MaskLen0 & 31) - 1)
6965 : (MaskIdx0 + MaskLen0 - 1),
6966 DL, GRLenVT),
6967 DAG.getConstant(MaskIdx0, DL, GRLenVT));
6968 }
6969
6970 // 6th pattern.
6971 // a = b | ((c & mask) << shamt), where all positions in b to be overwritten
6972 // by the incoming bits are known to be zero.
6973 // =>
6974 // a = BSTRINS b, c, shamt + MaskLen - 1, shamt
6975 //
6976 // Note that the 1st pattern is a special situation of the 6th, i.e. the 6th
6977 // pattern is more common than the 1st. So we put the 1st before the 6th in
6978 // order to match as many nodes as possible.
6979 ConstantSDNode *CNMask, *CNShamt;
6980 unsigned MaskIdx, MaskLen;
6981 if (N1.getOpcode() == ISD::SHL && N1.getOperand(0).getOpcode() == ISD::AND &&
6982 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
6983 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen) &&
6984 MaskIdx == 0 && (CNShamt = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
6985 CNShamt->getZExtValue() + MaskLen <= ValBits) {
6986 Shamt = CNShamt->getZExtValue();
6987 APInt ShMask(ValBits, CNMask->getZExtValue() << Shamt);
6988 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
6989 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 6\n");
6990 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
6991 N1.getOperand(0).getOperand(0),
6992 DAG.getConstant(Shamt + MaskLen - 1, DL, GRLenVT),
6993 DAG.getConstant(Shamt, DL, GRLenVT));
6994 }
6995 }
6996
6997 // 7th pattern.
6998 // a = b | ((c << shamt) & shifted_mask), where all positions in b to be
6999 // overwritten by the incoming bits are known to be zero.
7000 // =>
7001 // a = BSTRINS b, c, MaskIdx + MaskLen - 1, MaskIdx
7002 //
7003 // Similarly, the 7th pattern is more common than the 2nd. So we put the 2nd
7004 // before the 7th in order to match as many nodes as possible.
7005 if (N1.getOpcode() == ISD::AND &&
7006 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
7007 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen) &&
7008 N1.getOperand(0).getOpcode() == ISD::SHL &&
7009 (CNShamt = dyn_cast<ConstantSDNode>(N1.getOperand(0).getOperand(1))) &&
7010 CNShamt->getZExtValue() == MaskIdx) {
7011 APInt ShMask(ValBits, CNMask->getZExtValue());
7012 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
7013 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 7\n");
7014 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
7015 N1.getOperand(0).getOperand(0),
7016 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
7017 DAG.getConstant(MaskIdx, DL, GRLenVT));
7018 }
7019 }
7020
7021 // (or a, b) and (or b, a) are equivalent, so swap the operands and retry.
7022 if (!SwapAndRetried) {
7023 std::swap(N0, N1);
7024 SwapAndRetried = true;
7025 goto Retry;
7026 }
7027
7028 SwapAndRetried = false;
7029Retry2:
7030 // 8th pattern.
7031 // a = b | (c & shifted_mask), where all positions in b to be overwritten by
7032 // the incoming bits are known to be zero.
7033 // =>
7034 // a = BSTRINS b, c >> MaskIdx, MaskIdx + MaskLen - 1, MaskIdx
7035 //
7036 // Similarly, the 8th pattern is more common than the 4th and 5th patterns. So
7037 // we put it here in order to match as many nodes as possible or generate less
7038 // instructions.
7039 if (N1.getOpcode() == ISD::AND &&
7040 (CNMask = dyn_cast<ConstantSDNode>(N1.getOperand(1))) &&
7041 isShiftedMask_64(CNMask->getZExtValue(), MaskIdx, MaskLen)) {
7042 APInt ShMask(ValBits, CNMask->getZExtValue());
7043 if (ShMask.isSubsetOf(DAG.computeKnownBits(N0).Zero)) {
7044 LLVM_DEBUG(dbgs() << "Perform OR combine: match pattern 8\n");
7045 return DAG.getNode(LoongArchISD::BSTRINS, DL, ValTy, N0,
7046 DAG.getNode(ISD::SRL, DL, N1->getValueType(0),
7047 N1->getOperand(0),
7048 DAG.getConstant(MaskIdx, DL, GRLenVT)),
7049 DAG.getConstant(MaskIdx + MaskLen - 1, DL, GRLenVT),
7050 DAG.getConstant(MaskIdx, DL, GRLenVT));
7051 }
7052 }
7053 // Swap N0/N1 and retry.
7054 if (!SwapAndRetried) {
7055 std::swap(N0, N1);
7056 SwapAndRetried = true;
7057 goto Retry2;
7058 }
7059
7060 return SDValue();
7061}
7062
7063static bool checkValueWidth(SDValue V, ISD::LoadExtType &ExtType) {
7064 ExtType = ISD::NON_EXTLOAD;
7065
7066 switch (V.getNode()->getOpcode()) {
7067 case ISD::LOAD: {
7068 LoadSDNode *LoadNode = cast<LoadSDNode>(V.getNode());
7069 if ((LoadNode->getMemoryVT() == MVT::i8) ||
7070 (LoadNode->getMemoryVT() == MVT::i16)) {
7071 ExtType = LoadNode->getExtensionType();
7072 return true;
7073 }
7074 return false;
7075 }
7076 case ISD::AssertSext: {
7077 VTSDNode *TypeNode = cast<VTSDNode>(V.getNode()->getOperand(1));
7078 if ((TypeNode->getVT() == MVT::i8) || (TypeNode->getVT() == MVT::i16)) {
7079 ExtType = ISD::SEXTLOAD;
7080 return true;
7081 }
7082 return false;
7083 }
7084 case ISD::AssertZext: {
7085 VTSDNode *TypeNode = cast<VTSDNode>(V.getNode()->getOperand(1));
7086 if ((TypeNode->getVT() == MVT::i8) || (TypeNode->getVT() == MVT::i16)) {
7087 ExtType = ISD::ZEXTLOAD;
7088 return true;
7089 }
7090 return false;
7091 }
7092 default:
7093 return false;
7094 }
7095
7096 return false;
7097}
7098
7099// Fold a lane-mask extraction that is only compared against zero into one of
7100// all-lane check instructions, which write the results into fcc register, so
7101// BCEQZ could use it directly.
7102//
7103// (VMSKLTZ X) != 0 -> VSETNEZ.V X
7104// (VMSKLTZ X) == 0 -> VSETEQZ.V X
7105//
7106// This is valid when X is the result of a vector compare instruction,
7107// so each lane of X has to be all-ones or all-zeros.
7109 const SDLoc &DL, SelectionDAG &DAG,
7110 const LoongArchSubtarget &Subtarget) {
7111 if (CC != ISD::SETEQ && CC != ISD::SETNE)
7112 return SDValue();
7113 if (!isNullConstant(RHS))
7114 return SDValue();
7115
7116 unsigned MskOpc = LHS.getOpcode();
7117 if (MskOpc != LoongArchISD::VMSKLTZ && MskOpc != LoongArchISD::XVMSKLTZ)
7118 return SDValue();
7119 // Keeping the mask alive for another user would defeat the purpose.
7120 if (!LHS.hasOneUse())
7121 return SDValue();
7122
7123 SDValue Src = LHS.getOperand(0);
7124 EVT SrcVT = Src.getValueType();
7125 // Make sure every lane is all-ones or all-zeros.
7126 if (!SrcVT.isVector() ||
7127 DAG.ComputeNumSignBits(Src) != SrcVT.getScalarSizeInBits())
7128 return SDValue();
7129
7130 return DAG.getNode(CC == ISD::SETNE ? LoongArchISD::VANYNONZERO
7131 : LoongArchISD::VALLZERO,
7132 DL, Subtarget.getGRLenVT(), Src);
7133}
7134
7135// Eliminate redundant truncation and zero-extension nodes.
7136// * Case 1:
7137// +------------+ +------------+ +------------+
7138// | Input1 | | Input2 | | CC |
7139// +------------+ +------------+ +------------+
7140// | | |
7141// V V +----+
7142// +------------+ +------------+ |
7143// | TRUNCATE | | TRUNCATE | |
7144// +------------+ +------------+ |
7145// | | |
7146// V V |
7147// +------------+ +------------+ |
7148// | ZERO_EXT | | ZERO_EXT | |
7149// +------------+ +------------+ |
7150// | | |
7151// | +-------------+ |
7152// V V | |
7153// +----------------+ | |
7154// | AND | | |
7155// +----------------+ | |
7156// | | |
7157// +---------------+ | |
7158// | | |
7159// V V V
7160// +-------------+
7161// | CMP |
7162// +-------------+
7163// * Case 2:
7164// +------------+ +------------+ +-------------+ +------------+ +------------+
7165// | Input1 | | Input2 | | Constant -1 | | Constant 0 | | CC |
7166// +------------+ +------------+ +-------------+ +------------+ +------------+
7167// | | | | |
7168// V | | | |
7169// +------------+ | | | |
7170// | XOR |<---------------------+ | |
7171// +------------+ | | |
7172// | | | |
7173// V V +---------------+ |
7174// +------------+ +------------+ | |
7175// | TRUNCATE | | TRUNCATE | | +-------------------------+
7176// +------------+ +------------+ | |
7177// | | | |
7178// V V | |
7179// +------------+ +------------+ | |
7180// | ZERO_EXT | | ZERO_EXT | | |
7181// +------------+ +------------+ | |
7182// | | | |
7183// V V | |
7184// +----------------+ | |
7185// | AND | | |
7186// +----------------+ | |
7187// | | |
7188// +---------------+ | |
7189// | | |
7190// V V V
7191// +-------------+
7192// | CMP |
7193// +-------------+
7196 const LoongArchSubtarget &Subtarget) {
7197 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
7198
7199 if (N->getValueType(0) == Subtarget.getGRLenVT())
7200 if (SDValue V = foldVMskZeroTest(N->getOperand(0), N->getOperand(1), CC,
7201 SDLoc(N), DAG, Subtarget))
7202 return V;
7203
7204 SDNode *AndNode = N->getOperand(0).getNode();
7205 if (AndNode->getOpcode() != ISD::AND)
7206 return SDValue();
7207
7208 SDValue AndInputValue2 = AndNode->getOperand(1);
7209 if (AndInputValue2.getOpcode() != ISD::ZERO_EXTEND)
7210 return SDValue();
7211
7212 SDValue CmpInputValue = N->getOperand(1);
7213 SDValue AndInputValue1 = AndNode->getOperand(0);
7214 if (AndInputValue1.getOpcode() == ISD::XOR) {
7215 if (CC != ISD::SETEQ && CC != ISD::SETNE)
7216 return SDValue();
7217 ConstantSDNode *CN = dyn_cast<ConstantSDNode>(AndInputValue1.getOperand(1));
7218 if (!CN || !CN->isAllOnes())
7219 return SDValue();
7220 CN = dyn_cast<ConstantSDNode>(CmpInputValue);
7221 if (!CN || !CN->isZero())
7222 return SDValue();
7223 AndInputValue1 = AndInputValue1.getOperand(0);
7224 if (AndInputValue1.getOpcode() != ISD::ZERO_EXTEND)
7225 return SDValue();
7226 } else if (AndInputValue1.getOpcode() == ISD::ZERO_EXTEND) {
7227 if (AndInputValue2 != CmpInputValue)
7228 return SDValue();
7229 } else {
7230 return SDValue();
7231 }
7232
7233 SDValue TruncValue1 = AndInputValue1.getNode()->getOperand(0);
7234 if (TruncValue1.getOpcode() != ISD::TRUNCATE)
7235 return SDValue();
7236
7237 SDValue TruncValue2 = AndInputValue2.getNode()->getOperand(0);
7238 if (TruncValue2.getOpcode() != ISD::TRUNCATE)
7239 return SDValue();
7240
7241 SDValue TruncInputValue1 = TruncValue1.getNode()->getOperand(0);
7242 SDValue TruncInputValue2 = TruncValue2.getNode()->getOperand(0);
7243 ISD::LoadExtType ExtType1;
7244 ISD::LoadExtType ExtType2;
7245
7246 if (!checkValueWidth(TruncInputValue1, ExtType1) ||
7247 !checkValueWidth(TruncInputValue2, ExtType2))
7248 return SDValue();
7249
7250 if (TruncInputValue1->getValueType(0) != TruncInputValue2->getValueType(0) ||
7251 AndNode->getValueType(0) != TruncInputValue1->getValueType(0))
7252 return SDValue();
7253
7254 if ((ExtType2 != ISD::ZEXTLOAD) &&
7255 ((ExtType2 != ISD::SEXTLOAD) && (ExtType1 != ISD::SEXTLOAD)))
7256 return SDValue();
7257
7258 // These truncation and zero-extension nodes are not necessary, remove them.
7259 SDValue NewAnd = DAG.getNode(ISD::AND, SDLoc(N), AndNode->getValueType(0),
7260 TruncInputValue1, TruncInputValue2);
7261 SDValue NewSetCC =
7262 DAG.getSetCC(SDLoc(N), N->getValueType(0), NewAnd, TruncInputValue2, CC);
7263 DAG.ReplaceAllUsesWith(N, NewSetCC.getNode());
7264 return SDValue(N, 0);
7265}
7266
7267// Strip a single outer ISD::SIGN_EXTEND_INREG from \p V, if present, and
7268// return the inner value together with the narrow VT it was extending
7269// from. If no such node is present, returns \p V unchanged and an invalid
7270// EVT.
7271//
7272// i32 (and other sub-GRLen) arithmetic is legalized to operate on the full
7273// GRLen-width register, with a `sign_extend_inreg` re-normalizing the
7274// result back into the narrow type's range afterwards (see e.g. the
7275// `add i32` -> `add` + `sign_extend_inreg ..., i32` legalization). Any
7276// combine that reassociates such a binop must track and reapply this
7277// extension, otherwise the transformed code can produce a value whose
7278// high bits no longer match the narrow-type semantics.
7279static std::pair<SDValue, EVT> stripSignExtendInReg(SDValue V) {
7280 if (V.getOpcode() == ISD::SIGN_EXTEND_INREG)
7281 return {V.getOperand(0), cast<VTSDNode>(V.getOperand(1))->getVT()};
7282 return {V, EVT()};
7283}
7284
7285// Try to match \p BinV (after optionally stripping an outer
7286// sign_extend_inreg) as a supported binary operation that has \p X as one
7287// of its operands, returning the matched opcode, the other operand (the
7288// "delta"), and the narrow VT of the sign_extend_inreg that was stripped
7289// (invalid EVT if none was present).
7290//
7291// For commutative ops (add/or/xor), \p X may be either operand, since
7292// `binop(X, Y) == binop(Y, X)` and the identity element (0) works on
7293// either side.
7294//
7295// For `sub`, the operation is NOT commutative: `sub(X, Y) != sub(Y, X)`.
7296// Only `sub(X, Y)` (i.e. \p X is the *minuend*, the first operand) can be
7297// rewritten using the identity `X - 0 == X`. If \p X were the *subtrahend*
7298// (second operand, i.e. the pattern is actually `sub(Y, X)`), there is no
7299// way to express `cond ? (Y - X) : X` (or the symmetric case) as
7300// `X op (select ...)` without introducing an extra negation, so that case
7301// must be rejected instead of "optimized" into worse code.
7302static std::tuple<unsigned, SDValue, EVT>
7304 auto [Inner, ExtVT] = stripSignExtendInReg(BinV);
7305
7306 unsigned Opc = Inner.getOpcode();
7307 switch (Opc) {
7308 case ISD::ADD:
7309 case ISD::OR:
7310 case ISD::XOR:
7311 if (Inner.getOperand(0) == X)
7312 return {Opc, Inner.getOperand(1), ExtVT};
7313 if (Inner.getOperand(1) == X)
7314 return {Opc, Inner.getOperand(0), ExtVT};
7315 return {0, SDValue(), EVT()};
7316 case ISD::SUB:
7317 // Only accept X as the minuend (first operand); see comment above.
7318 if (Inner.getOperand(0) == X)
7319 return {Opc, Inner.getOperand(1), ExtVT};
7320 return {0, SDValue(), EVT()};
7321 default:
7322 return {0, SDValue(), EVT()};
7323 }
7324}
7325
7326// Try to combine:
7327// select cond, binop(X, Y), X -> binop X, (select cond, Y, 0)
7328// select cond, X, binop(X, Y) -> binop X, (select cond, 0, Y)
7329// for binop in {add, or, xor, sub}, where 0 is the identity element of the
7330// respective operation, additionally handling the common legalized form
7331// where the binop result is wrapped in a `sign_extend_inreg` (as happens
7332// for sub-GRLen types such as i32 on a 64-bit GRLen target). See
7333// matchBinOpWithSharedOperand() for the restrictions applied to
7334// non-commutative operations (currently only `sub`).
7337 const LoongArchSubtarget &Subtarget) {
7338 if (DCI.isBeforeLegalizeOps())
7339 return SDValue();
7340
7341 EVT VT = N->getValueType(0);
7342 // Restrict to the scalar GRLen integer type that maskeqz/masknez operate
7343 // on; this also naturally excludes float and vector selects.
7344 if (VT != Subtarget.getGRLenVT())
7345 return SDValue();
7346
7347 SDValue Cond = N->getOperand(0);
7348 SDValue TrueV = N->getOperand(1);
7349 SDValue FalseV = N->getOperand(2);
7350 SDLoc DL(N);
7351
7352 auto TryFold = [&](SDValue BinV, SDValue SharedV,
7353 bool BinIsTrueArm) -> SDValue {
7354 auto [Opc, Delta, ExtVT] = matchBinOpWithSharedOperand(BinV, SharedV);
7355 if (!Opc)
7356 return SDValue();
7357
7358 // Avoid infinite combine loops: bail out if Delta is trivially the
7359 // same node we would otherwise be selecting on (shouldn't normally
7360 // happen, but guards against degenerate/self-referential IR).
7361 if (Delta.getNode() == N)
7362 return SDValue();
7363
7364 SDValue Zero = DAG.getConstant(0, DL, VT);
7365 SDValue NewSel = BinIsTrueArm ? DAG.getSelect(DL, VT, Cond, Delta, Zero)
7366 : DAG.getSelect(DL, VT, Cond, Zero, Delta);
7367 SDValue NewBin = DAG.getNode(Opc, DL, VT, SharedV, NewSel);
7368
7369 // If the original binop result was normalized back into a narrower
7370 // type via sign_extend_inreg (e.g. i32 arithmetic on a 64-bit GRLen
7371 // target), the new binop must be re-normalized the same way: SharedV
7372 // is already known-sign-extended for that narrow type, but NewSel
7373 // (Delta or 0, selected) combined with SharedV via Opc can still
7374 // produce a 64-bit result whose high bits don't match the narrow
7375 // type's sign-extended representation.
7376 if (ExtVT != EVT())
7377 NewBin = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, NewBin,
7378 DAG.getValueType(ExtVT));
7379
7380 return NewBin;
7381 };
7382
7383 if (SDValue R = TryFold(TrueV, FalseV, /*BinIsTrueArm=*/true))
7384 return R;
7385 if (SDValue R = TryFold(FalseV, TrueV, /*BinIsTrueArm=*/false))
7386 return R;
7387
7388 return SDValue();
7389}
7390
7391// Combine (loongarch_bitrev_w (loongarch_revb_2w X)) to loongarch_bitrev_4b.
7394 const LoongArchSubtarget &Subtarget) {
7395 if (DCI.isBeforeLegalizeOps())
7396 return SDValue();
7397
7398 SDValue Src = N->getOperand(0);
7399 if (Src.getOpcode() != LoongArchISD::REVB_2W)
7400 return SDValue();
7401
7402 return DAG.getNode(LoongArchISD::BITREV_4B, SDLoc(N), N->getValueType(0),
7403 Src.getOperand(0));
7404}
7405
7406// Perform common combines for BR_CC and SELECT_CC conditions.
7407static bool combine_CC(SDValue &LHS, SDValue &RHS, SDValue &CC, const SDLoc &DL,
7408 SelectionDAG &DAG, const LoongArchSubtarget &Subtarget) {
7409 ISD::CondCode CCVal = cast<CondCodeSDNode>(CC)->get();
7410
7411 // As far as arithmetic right shift always saves the sign,
7412 // shift can be omitted.
7413 // Fold setlt (sra X, N), 0 -> setlt X, 0 and
7414 // setge (sra X, N), 0 -> setge X, 0
7415 if (isNullConstant(RHS) && (CCVal == ISD::SETGE || CCVal == ISD::SETLT) &&
7416 LHS.getOpcode() == ISD::SRA) {
7417 LHS = LHS.getOperand(0);
7418 return true;
7419 }
7420
7421 if (!ISD::isIntEqualitySetCC(CCVal))
7422 return false;
7423
7424 // Fold ((setlt X, Y), 0, ne) -> (X, Y, lt)
7425 // Sometimes the setcc is introduced after br_cc/select_cc has been formed.
7426 if (LHS.getOpcode() == ISD::SETCC && isNullConstant(RHS) &&
7427 LHS.getOperand(0).getValueType() == Subtarget.getGRLenVT()) {
7428 // If we're looking for eq 0 instead of ne 0, we need to invert the
7429 // condition.
7430 bool Invert = CCVal == ISD::SETEQ;
7431 CCVal = cast<CondCodeSDNode>(LHS.getOperand(2))->get();
7432 if (Invert)
7433 CCVal = ISD::getSetCCInverse(CCVal, LHS.getValueType());
7434
7435 RHS = LHS.getOperand(1);
7436 LHS = LHS.getOperand(0);
7437 translateSetCCForBranch(DL, LHS, RHS, CCVal, DAG);
7438
7439 CC = DAG.getCondCode(CCVal);
7440 return true;
7441 }
7442
7443 // Fold ((srl (and X, 1<<C), C), 0, eq/ne) -> ((shl X, GRLen-1-C), 0, ge/lt)
7444 if (isNullConstant(RHS) && LHS.getOpcode() == ISD::SRL && LHS.hasOneUse() &&
7445 LHS.getOperand(1).getOpcode() == ISD::Constant) {
7446 SDValue LHS0 = LHS.getOperand(0);
7447 if (LHS0.getOpcode() == ISD::AND &&
7448 LHS0.getOperand(1).getOpcode() == ISD::Constant) {
7449 uint64_t Mask = LHS0.getConstantOperandVal(1);
7450 uint64_t ShAmt = LHS.getConstantOperandVal(1);
7451 if (isPowerOf2_64(Mask) && Log2_64(Mask) == ShAmt) {
7452 CCVal = CCVal == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
7453 CC = DAG.getCondCode(CCVal);
7454
7455 ShAmt = LHS.getValueSizeInBits() - 1 - ShAmt;
7456 LHS = LHS0.getOperand(0);
7457 if (ShAmt != 0)
7458 LHS =
7459 DAG.getNode(ISD::SHL, DL, LHS.getValueType(), LHS0.getOperand(0),
7460 DAG.getConstant(ShAmt, DL, LHS.getValueType()));
7461 return true;
7462 }
7463 }
7464 }
7465
7466 // (X, 1, setne) -> (X, 0, seteq) if we can prove X is 0/1.
7467 // This can occur when legalizing some floating point comparisons.
7468 APInt Mask = APInt::getBitsSetFrom(LHS.getValueSizeInBits(), 1);
7469 if (isOneConstant(RHS) && DAG.MaskedValueIsZero(LHS, Mask)) {
7470 CCVal = ISD::getSetCCInverse(CCVal, LHS.getValueType());
7471 CC = DAG.getCondCode(CCVal);
7472 RHS = DAG.getConstant(0, DL, LHS.getValueType());
7473 return true;
7474 }
7475
7476 // Fold ((shl (extract_vector_elt X, I), GRLen - EleBits)), 0, eq/ne) ->
7477 // ((extract_vector_elt X, I), 0, eq/ne)
7478 if (isNullConstant(RHS) && (CCVal == ISD::SETEQ || CCVal == ISD::SETNE) &&
7479 LHS.getOpcode() == ISD::SHL && LHS.hasOneUse() &&
7480 isa<ConstantSDNode>(LHS.getOperand(1))) {
7481 SDValue Ext = LHS.getOperand(0);
7482 unsigned Sht = LHS.getConstantOperandVal(1);
7483 if (Ext.getOpcode() == ISD::EXTRACT_VECTOR_ELT) {
7484 SDValue Vec = Ext.getOperand(0);
7485 unsigned EleBits = Vec.getScalarValueSizeInBits();
7486 if ((EleBits + Sht) == Subtarget.getGRLen()) {
7487 LHS = Ext;
7488 return true;
7489 }
7490 }
7491 }
7492
7493 // Fold (C1, C2, cond) -> (0, 0, seteq/setne)
7495 const LoongArchTargetLowering *TLI = Subtarget.getTargetLowering();
7496 EVT VT = LHS.getValueType();
7497 EVT SetCCResVT =
7498 TLI->getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
7499 if (SDValue Folded = DAG.FoldSetCC(SetCCResVT, LHS, RHS, CCVal, DL)) {
7500 LHS = DAG.getConstant(0, DL, VT);
7501 RHS = DAG.getConstant(0, DL, VT);
7502 CC = DAG.getCondCode(!isNullConstant(Folded) ? ISD::SETEQ : ISD::SETNE);
7503 return true;
7504 }
7505 }
7506
7507 return false;
7508}
7509
7512 const LoongArchSubtarget &Subtarget) {
7513 SDValue LHS = N->getOperand(1);
7514 SDValue RHS = N->getOperand(2);
7515 SDValue CC = N->getOperand(3);
7516 SDLoc DL(N);
7517
7518 // CC was folded into V, so always return ISD::SETNE is fine.
7520 DL, DAG, Subtarget))
7521 return DAG.getNode(LoongArchISD::BR_CC, DL, N->getValueType(0),
7522 N->getOperand(0), V,
7523 DAG.getConstant(0, DL, Subtarget.getGRLenVT()),
7524 DAG.getCondCode(ISD::SETNE), N->getOperand(4));
7525
7526 if (combine_CC(LHS, RHS, CC, DL, DAG, Subtarget))
7527 return DAG.getNode(LoongArchISD::BR_CC, DL, N->getValueType(0),
7528 N->getOperand(0), LHS, RHS, CC, N->getOperand(4));
7529
7530 return SDValue();
7531}
7532
7535 const LoongArchSubtarget &Subtarget) {
7536 // Transform
7537 SDValue LHS = N->getOperand(0);
7538 SDValue RHS = N->getOperand(1);
7539 SDValue CC = N->getOperand(2);
7540 ISD::CondCode CCVal = cast<CondCodeSDNode>(CC)->get();
7541 SDValue TrueV = N->getOperand(3);
7542 SDValue FalseV = N->getOperand(4);
7543 SDLoc DL(N);
7544 EVT VT = N->getValueType(0);
7545
7546 // If the True and False values are the same, we don't need a select_cc.
7547 if (TrueV == FalseV)
7548 return TrueV;
7549
7550 // (select (x < 0), y, z) -> x >> (GRLEN - 1) & (y - z) + z
7551 // (select (x >= 0), y, z) -> x >> (GRLEN - 1) & (z - y) + y
7552 if (isa<ConstantSDNode>(TrueV) && isa<ConstantSDNode>(FalseV) &&
7554 (CCVal == ISD::CondCode::SETLT || CCVal == ISD::CondCode::SETGE)) {
7555 if (CCVal == ISD::CondCode::SETGE)
7556 std::swap(TrueV, FalseV);
7557
7558 int64_t TrueSImm = cast<ConstantSDNode>(TrueV)->getSExtValue();
7559 int64_t FalseSImm = cast<ConstantSDNode>(FalseV)->getSExtValue();
7560 // Only handle simm12, if it is not in this range, it can be considered as
7561 // register.
7562 if (isInt<12>(TrueSImm) && isInt<12>(FalseSImm) &&
7563 isInt<12>(TrueSImm - FalseSImm)) {
7564 SDValue SRA =
7565 DAG.getNode(ISD::SRA, DL, VT, LHS,
7566 DAG.getConstant(Subtarget.getGRLen() - 1, DL, VT));
7567 SDValue AND =
7568 DAG.getNode(ISD::AND, DL, VT, SRA,
7569 DAG.getSignedConstant(TrueSImm - FalseSImm, DL, VT));
7570 return DAG.getNode(ISD::ADD, DL, VT, AND, FalseV);
7571 }
7572
7573 if (CCVal == ISD::CondCode::SETGE)
7574 std::swap(TrueV, FalseV);
7575 }
7576
7577 if (combine_CC(LHS, RHS, CC, DL, DAG, Subtarget))
7578 return DAG.getNode(LoongArchISD::SELECT_CC, DL, N->getValueType(0),
7579 {LHS, RHS, CC, TrueV, FalseV});
7580
7581 return SDValue();
7582}
7583
7584template <unsigned N>
7586 SelectionDAG &DAG,
7587 const LoongArchSubtarget &Subtarget,
7588 bool IsSigned = false) {
7589 SDLoc DL(Node);
7590 auto *CImm = cast<ConstantSDNode>(Node->getOperand(ImmOp));
7591 // Check the ImmArg.
7592 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
7593 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
7594 DAG.getContext()->emitError(Node->getOperationName(0) +
7595 ": argument out of range.");
7596 return DAG.getNode(ISD::UNDEF, DL, Subtarget.getGRLenVT());
7597 }
7598 return DAG.getConstant(CImm->getZExtValue(), DL, Subtarget.getGRLenVT());
7599}
7600
7601template <unsigned N>
7602static SDValue lowerVectorSplatImm(SDNode *Node, unsigned ImmOp,
7603 SelectionDAG &DAG, bool IsSigned = false) {
7604 SDLoc DL(Node);
7605 EVT ResTy = Node->getValueType(0);
7606 auto *CImm = cast<ConstantSDNode>(Node->getOperand(ImmOp));
7607
7608 // Check the ImmArg.
7609 if ((IsSigned && !isInt<N>(CImm->getSExtValue())) ||
7610 (!IsSigned && !isUInt<N>(CImm->getZExtValue()))) {
7611 DAG.getContext()->emitError(Node->getOperationName(0) +
7612 ": argument out of range.");
7613 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7614 }
7615 return DAG.getConstant(
7617 IsSigned ? CImm->getSExtValue() : CImm->getZExtValue(), IsSigned),
7618 DL, ResTy);
7619}
7620
7622 SDLoc DL(Node);
7623 EVT ResTy = Node->getValueType(0);
7624 SDValue Vec = Node->getOperand(2);
7625 SDValue Mask = DAG.getConstant(Vec.getScalarValueSizeInBits() - 1, DL, ResTy);
7626 return DAG.getNode(ISD::AND, DL, ResTy, Vec, Mask);
7627}
7628
7630 SDLoc DL(Node);
7631 EVT ResTy = Node->getValueType(0);
7632 SDValue One = DAG.getConstant(1, DL, ResTy);
7633 SDValue Bit =
7634 DAG.getNode(ISD::SHL, DL, ResTy, One, truncateVecElts(Node, DAG));
7635
7636 return DAG.getNode(ISD::AND, DL, ResTy, Node->getOperand(1),
7637 DAG.getNOT(DL, Bit, ResTy));
7638}
7639
7640template <unsigned N>
7642 SDLoc DL(Node);
7643 EVT ResTy = Node->getValueType(0);
7644 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7645 // Check the unsigned ImmArg.
7646 if (!isUInt<N>(CImm->getZExtValue())) {
7647 DAG.getContext()->emitError(Node->getOperationName(0) +
7648 ": argument out of range.");
7649 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7650 }
7651
7652 APInt BitImm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7653 SDValue Mask = DAG.getConstant(~BitImm, DL, ResTy);
7654
7655 return DAG.getNode(ISD::AND, DL, ResTy, Node->getOperand(1), Mask);
7656}
7657
7658template <unsigned N>
7660 SDLoc DL(Node);
7661 EVT ResTy = Node->getValueType(0);
7662 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7663 // Check the unsigned ImmArg.
7664 if (!isUInt<N>(CImm->getZExtValue())) {
7665 DAG.getContext()->emitError(Node->getOperationName(0) +
7666 ": argument out of range.");
7667 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7668 }
7669
7670 APInt Imm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7671 SDValue BitImm = DAG.getConstant(Imm, DL, ResTy);
7672 return DAG.getNode(ISD::OR, DL, ResTy, Node->getOperand(1), BitImm);
7673}
7674
7675template <unsigned N>
7677 SDLoc DL(Node);
7678 EVT ResTy = Node->getValueType(0);
7679 auto *CImm = cast<ConstantSDNode>(Node->getOperand(2));
7680 // Check the unsigned ImmArg.
7681 if (!isUInt<N>(CImm->getZExtValue())) {
7682 DAG.getContext()->emitError(Node->getOperationName(0) +
7683 ": argument out of range.");
7684 return DAG.getNode(ISD::UNDEF, DL, ResTy);
7685 }
7686
7687 APInt Imm = APInt(ResTy.getScalarSizeInBits(), 1) << CImm->getAPIntValue();
7688 SDValue BitImm = DAG.getConstant(Imm, DL, ResTy);
7689 return DAG.getNode(ISD::XOR, DL, ResTy, Node->getOperand(1), BitImm);
7690}
7691
7692template <unsigned W>
7694 unsigned ResOp) {
7695 unsigned Imm = N->getConstantOperandVal(2);
7696 if (!isUInt<W>(Imm)) {
7697 const StringRef ErrorMsg = "argument out of range";
7698 DAG.getContext()->emitError(N->getOperationName(0) + ": " + ErrorMsg + ".");
7699 return DAG.getUNDEF(N->getValueType(0));
7700 }
7701 SDLoc DL(N);
7702 SDValue Vec = N->getOperand(1);
7703 SDValue Idx = DAG.getConstant(Imm, DL, MVT::i32);
7705 return DAG.getNode(ResOp, DL, N->getValueType(0), Vec, Idx, EltVT);
7706}
7707
7708static SDValue
7711 const LoongArchSubtarget &Subtarget) {
7712 SDLoc DL(N);
7713 switch (N->getConstantOperandVal(0)) {
7714 default:
7715 break;
7716 case Intrinsic::loongarch_lsx_vadd_b:
7717 case Intrinsic::loongarch_lsx_vadd_h:
7718 case Intrinsic::loongarch_lsx_vadd_w:
7719 case Intrinsic::loongarch_lsx_vadd_d:
7720 case Intrinsic::loongarch_lasx_xvadd_b:
7721 case Intrinsic::loongarch_lasx_xvadd_h:
7722 case Intrinsic::loongarch_lasx_xvadd_w:
7723 case Intrinsic::loongarch_lasx_xvadd_d:
7724 return DAG.getNode(ISD::ADD, DL, N->getValueType(0), N->getOperand(1),
7725 N->getOperand(2));
7726 case Intrinsic::loongarch_lsx_vaddi_bu:
7727 case Intrinsic::loongarch_lsx_vaddi_hu:
7728 case Intrinsic::loongarch_lsx_vaddi_wu:
7729 case Intrinsic::loongarch_lsx_vaddi_du:
7730 case Intrinsic::loongarch_lasx_xvaddi_bu:
7731 case Intrinsic::loongarch_lasx_xvaddi_hu:
7732 case Intrinsic::loongarch_lasx_xvaddi_wu:
7733 case Intrinsic::loongarch_lasx_xvaddi_du:
7734 return DAG.getNode(ISD::ADD, DL, N->getValueType(0), N->getOperand(1),
7735 lowerVectorSplatImm<5>(N, 2, DAG));
7736 case Intrinsic::loongarch_lsx_vsub_b:
7737 case Intrinsic::loongarch_lsx_vsub_h:
7738 case Intrinsic::loongarch_lsx_vsub_w:
7739 case Intrinsic::loongarch_lsx_vsub_d:
7740 case Intrinsic::loongarch_lasx_xvsub_b:
7741 case Intrinsic::loongarch_lasx_xvsub_h:
7742 case Intrinsic::loongarch_lasx_xvsub_w:
7743 case Intrinsic::loongarch_lasx_xvsub_d:
7744 return DAG.getNode(ISD::SUB, DL, N->getValueType(0), N->getOperand(1),
7745 N->getOperand(2));
7746 case Intrinsic::loongarch_lsx_vsubi_bu:
7747 case Intrinsic::loongarch_lsx_vsubi_hu:
7748 case Intrinsic::loongarch_lsx_vsubi_wu:
7749 case Intrinsic::loongarch_lsx_vsubi_du:
7750 case Intrinsic::loongarch_lasx_xvsubi_bu:
7751 case Intrinsic::loongarch_lasx_xvsubi_hu:
7752 case Intrinsic::loongarch_lasx_xvsubi_wu:
7753 case Intrinsic::loongarch_lasx_xvsubi_du:
7754 return DAG.getNode(ISD::SUB, DL, N->getValueType(0), N->getOperand(1),
7755 lowerVectorSplatImm<5>(N, 2, DAG));
7756 case Intrinsic::loongarch_lsx_vneg_b:
7757 case Intrinsic::loongarch_lsx_vneg_h:
7758 case Intrinsic::loongarch_lsx_vneg_w:
7759 case Intrinsic::loongarch_lsx_vneg_d:
7760 case Intrinsic::loongarch_lasx_xvneg_b:
7761 case Intrinsic::loongarch_lasx_xvneg_h:
7762 case Intrinsic::loongarch_lasx_xvneg_w:
7763 case Intrinsic::loongarch_lasx_xvneg_d:
7764 return DAG.getNode(
7765 ISD::SUB, DL, N->getValueType(0),
7766 DAG.getConstant(
7767 APInt(N->getValueType(0).getScalarType().getSizeInBits(), 0,
7768 /*isSigned=*/true),
7769 SDLoc(N), N->getValueType(0)),
7770 N->getOperand(1));
7771 case Intrinsic::loongarch_lsx_vmax_b:
7772 case Intrinsic::loongarch_lsx_vmax_h:
7773 case Intrinsic::loongarch_lsx_vmax_w:
7774 case Intrinsic::loongarch_lsx_vmax_d:
7775 case Intrinsic::loongarch_lasx_xvmax_b:
7776 case Intrinsic::loongarch_lasx_xvmax_h:
7777 case Intrinsic::loongarch_lasx_xvmax_w:
7778 case Intrinsic::loongarch_lasx_xvmax_d:
7779 return DAG.getNode(ISD::SMAX, DL, N->getValueType(0), N->getOperand(1),
7780 N->getOperand(2));
7781 case Intrinsic::loongarch_lsx_vmax_bu:
7782 case Intrinsic::loongarch_lsx_vmax_hu:
7783 case Intrinsic::loongarch_lsx_vmax_wu:
7784 case Intrinsic::loongarch_lsx_vmax_du:
7785 case Intrinsic::loongarch_lasx_xvmax_bu:
7786 case Intrinsic::loongarch_lasx_xvmax_hu:
7787 case Intrinsic::loongarch_lasx_xvmax_wu:
7788 case Intrinsic::loongarch_lasx_xvmax_du:
7789 return DAG.getNode(ISD::UMAX, DL, N->getValueType(0), N->getOperand(1),
7790 N->getOperand(2));
7791 case Intrinsic::loongarch_lsx_vmaxi_b:
7792 case Intrinsic::loongarch_lsx_vmaxi_h:
7793 case Intrinsic::loongarch_lsx_vmaxi_w:
7794 case Intrinsic::loongarch_lsx_vmaxi_d:
7795 case Intrinsic::loongarch_lasx_xvmaxi_b:
7796 case Intrinsic::loongarch_lasx_xvmaxi_h:
7797 case Intrinsic::loongarch_lasx_xvmaxi_w:
7798 case Intrinsic::loongarch_lasx_xvmaxi_d:
7799 return DAG.getNode(ISD::SMAX, DL, N->getValueType(0), N->getOperand(1),
7800 lowerVectorSplatImm<5>(N, 2, DAG, /*IsSigned=*/true));
7801 case Intrinsic::loongarch_lsx_vmaxi_bu:
7802 case Intrinsic::loongarch_lsx_vmaxi_hu:
7803 case Intrinsic::loongarch_lsx_vmaxi_wu:
7804 case Intrinsic::loongarch_lsx_vmaxi_du:
7805 case Intrinsic::loongarch_lasx_xvmaxi_bu:
7806 case Intrinsic::loongarch_lasx_xvmaxi_hu:
7807 case Intrinsic::loongarch_lasx_xvmaxi_wu:
7808 case Intrinsic::loongarch_lasx_xvmaxi_du:
7809 return DAG.getNode(ISD::UMAX, DL, N->getValueType(0), N->getOperand(1),
7810 lowerVectorSplatImm<5>(N, 2, DAG));
7811 case Intrinsic::loongarch_lsx_vmin_b:
7812 case Intrinsic::loongarch_lsx_vmin_h:
7813 case Intrinsic::loongarch_lsx_vmin_w:
7814 case Intrinsic::loongarch_lsx_vmin_d:
7815 case Intrinsic::loongarch_lasx_xvmin_b:
7816 case Intrinsic::loongarch_lasx_xvmin_h:
7817 case Intrinsic::loongarch_lasx_xvmin_w:
7818 case Intrinsic::loongarch_lasx_xvmin_d:
7819 return DAG.getNode(ISD::SMIN, DL, N->getValueType(0), N->getOperand(1),
7820 N->getOperand(2));
7821 case Intrinsic::loongarch_lsx_vmin_bu:
7822 case Intrinsic::loongarch_lsx_vmin_hu:
7823 case Intrinsic::loongarch_lsx_vmin_wu:
7824 case Intrinsic::loongarch_lsx_vmin_du:
7825 case Intrinsic::loongarch_lasx_xvmin_bu:
7826 case Intrinsic::loongarch_lasx_xvmin_hu:
7827 case Intrinsic::loongarch_lasx_xvmin_wu:
7828 case Intrinsic::loongarch_lasx_xvmin_du:
7829 return DAG.getNode(ISD::UMIN, DL, N->getValueType(0), N->getOperand(1),
7830 N->getOperand(2));
7831 case Intrinsic::loongarch_lsx_vmini_b:
7832 case Intrinsic::loongarch_lsx_vmini_h:
7833 case Intrinsic::loongarch_lsx_vmini_w:
7834 case Intrinsic::loongarch_lsx_vmini_d:
7835 case Intrinsic::loongarch_lasx_xvmini_b:
7836 case Intrinsic::loongarch_lasx_xvmini_h:
7837 case Intrinsic::loongarch_lasx_xvmini_w:
7838 case Intrinsic::loongarch_lasx_xvmini_d:
7839 return DAG.getNode(ISD::SMIN, DL, N->getValueType(0), N->getOperand(1),
7840 lowerVectorSplatImm<5>(N, 2, DAG, /*IsSigned=*/true));
7841 case Intrinsic::loongarch_lsx_vmini_bu:
7842 case Intrinsic::loongarch_lsx_vmini_hu:
7843 case Intrinsic::loongarch_lsx_vmini_wu:
7844 case Intrinsic::loongarch_lsx_vmini_du:
7845 case Intrinsic::loongarch_lasx_xvmini_bu:
7846 case Intrinsic::loongarch_lasx_xvmini_hu:
7847 case Intrinsic::loongarch_lasx_xvmini_wu:
7848 case Intrinsic::loongarch_lasx_xvmini_du:
7849 return DAG.getNode(ISD::UMIN, DL, N->getValueType(0), N->getOperand(1),
7850 lowerVectorSplatImm<5>(N, 2, DAG));
7851 case Intrinsic::loongarch_lsx_vmul_b:
7852 case Intrinsic::loongarch_lsx_vmul_h:
7853 case Intrinsic::loongarch_lsx_vmul_w:
7854 case Intrinsic::loongarch_lsx_vmul_d:
7855 case Intrinsic::loongarch_lasx_xvmul_b:
7856 case Intrinsic::loongarch_lasx_xvmul_h:
7857 case Intrinsic::loongarch_lasx_xvmul_w:
7858 case Intrinsic::loongarch_lasx_xvmul_d:
7859 return DAG.getNode(ISD::MUL, DL, N->getValueType(0), N->getOperand(1),
7860 N->getOperand(2));
7861 case Intrinsic::loongarch_lsx_vmadd_b:
7862 case Intrinsic::loongarch_lsx_vmadd_h:
7863 case Intrinsic::loongarch_lsx_vmadd_w:
7864 case Intrinsic::loongarch_lsx_vmadd_d:
7865 case Intrinsic::loongarch_lasx_xvmadd_b:
7866 case Intrinsic::loongarch_lasx_xvmadd_h:
7867 case Intrinsic::loongarch_lasx_xvmadd_w:
7868 case Intrinsic::loongarch_lasx_xvmadd_d: {
7869 EVT ResTy = N->getValueType(0);
7870 return DAG.getNode(ISD::ADD, SDLoc(N), ResTy, N->getOperand(1),
7871 DAG.getNode(ISD::MUL, SDLoc(N), ResTy, N->getOperand(2),
7872 N->getOperand(3)));
7873 }
7874 case Intrinsic::loongarch_lsx_vmsub_b:
7875 case Intrinsic::loongarch_lsx_vmsub_h:
7876 case Intrinsic::loongarch_lsx_vmsub_w:
7877 case Intrinsic::loongarch_lsx_vmsub_d:
7878 case Intrinsic::loongarch_lasx_xvmsub_b:
7879 case Intrinsic::loongarch_lasx_xvmsub_h:
7880 case Intrinsic::loongarch_lasx_xvmsub_w:
7881 case Intrinsic::loongarch_lasx_xvmsub_d: {
7882 EVT ResTy = N->getValueType(0);
7883 return DAG.getNode(ISD::SUB, SDLoc(N), ResTy, N->getOperand(1),
7884 DAG.getNode(ISD::MUL, SDLoc(N), ResTy, N->getOperand(2),
7885 N->getOperand(3)));
7886 }
7887 case Intrinsic::loongarch_lsx_vdiv_b:
7888 case Intrinsic::loongarch_lsx_vdiv_h:
7889 case Intrinsic::loongarch_lsx_vdiv_w:
7890 case Intrinsic::loongarch_lsx_vdiv_d:
7891 case Intrinsic::loongarch_lasx_xvdiv_b:
7892 case Intrinsic::loongarch_lasx_xvdiv_h:
7893 case Intrinsic::loongarch_lasx_xvdiv_w:
7894 case Intrinsic::loongarch_lasx_xvdiv_d:
7895 return DAG.getNode(ISD::SDIV, DL, N->getValueType(0), N->getOperand(1),
7896 N->getOperand(2));
7897 case Intrinsic::loongarch_lsx_vdiv_bu:
7898 case Intrinsic::loongarch_lsx_vdiv_hu:
7899 case Intrinsic::loongarch_lsx_vdiv_wu:
7900 case Intrinsic::loongarch_lsx_vdiv_du:
7901 case Intrinsic::loongarch_lasx_xvdiv_bu:
7902 case Intrinsic::loongarch_lasx_xvdiv_hu:
7903 case Intrinsic::loongarch_lasx_xvdiv_wu:
7904 case Intrinsic::loongarch_lasx_xvdiv_du:
7905 return DAG.getNode(ISD::UDIV, DL, N->getValueType(0), N->getOperand(1),
7906 N->getOperand(2));
7907 case Intrinsic::loongarch_lsx_vmod_b:
7908 case Intrinsic::loongarch_lsx_vmod_h:
7909 case Intrinsic::loongarch_lsx_vmod_w:
7910 case Intrinsic::loongarch_lsx_vmod_d:
7911 case Intrinsic::loongarch_lasx_xvmod_b:
7912 case Intrinsic::loongarch_lasx_xvmod_h:
7913 case Intrinsic::loongarch_lasx_xvmod_w:
7914 case Intrinsic::loongarch_lasx_xvmod_d:
7915 return DAG.getNode(ISD::SREM, DL, N->getValueType(0), N->getOperand(1),
7916 N->getOperand(2));
7917 case Intrinsic::loongarch_lsx_vmod_bu:
7918 case Intrinsic::loongarch_lsx_vmod_hu:
7919 case Intrinsic::loongarch_lsx_vmod_wu:
7920 case Intrinsic::loongarch_lsx_vmod_du:
7921 case Intrinsic::loongarch_lasx_xvmod_bu:
7922 case Intrinsic::loongarch_lasx_xvmod_hu:
7923 case Intrinsic::loongarch_lasx_xvmod_wu:
7924 case Intrinsic::loongarch_lasx_xvmod_du:
7925 return DAG.getNode(ISD::UREM, DL, N->getValueType(0), N->getOperand(1),
7926 N->getOperand(2));
7927 case Intrinsic::loongarch_lsx_vand_v:
7928 case Intrinsic::loongarch_lasx_xvand_v:
7929 return DAG.getNode(ISD::AND, DL, N->getValueType(0), N->getOperand(1),
7930 N->getOperand(2));
7931 case Intrinsic::loongarch_lsx_vor_v:
7932 case Intrinsic::loongarch_lasx_xvor_v:
7933 return DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7934 N->getOperand(2));
7935 case Intrinsic::loongarch_lsx_vxor_v:
7936 case Intrinsic::loongarch_lasx_xvxor_v:
7937 return DAG.getNode(ISD::XOR, DL, N->getValueType(0), N->getOperand(1),
7938 N->getOperand(2));
7939 case Intrinsic::loongarch_lsx_vnor_v:
7940 case Intrinsic::loongarch_lasx_xvnor_v: {
7941 SDValue Res = DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7942 N->getOperand(2));
7943 return DAG.getNOT(DL, Res, Res->getValueType(0));
7944 }
7945 case Intrinsic::loongarch_lsx_vandi_b:
7946 case Intrinsic::loongarch_lasx_xvandi_b:
7947 return DAG.getNode(ISD::AND, DL, N->getValueType(0), N->getOperand(1),
7948 lowerVectorSplatImm<8>(N, 2, DAG));
7949 case Intrinsic::loongarch_lsx_vori_b:
7950 case Intrinsic::loongarch_lasx_xvori_b:
7951 return DAG.getNode(ISD::OR, DL, N->getValueType(0), N->getOperand(1),
7952 lowerVectorSplatImm<8>(N, 2, DAG));
7953 case Intrinsic::loongarch_lsx_vxori_b:
7954 case Intrinsic::loongarch_lasx_xvxori_b:
7955 return DAG.getNode(ISD::XOR, DL, N->getValueType(0), N->getOperand(1),
7956 lowerVectorSplatImm<8>(N, 2, DAG));
7957 case Intrinsic::loongarch_lsx_vsll_b:
7958 case Intrinsic::loongarch_lsx_vsll_h:
7959 case Intrinsic::loongarch_lsx_vsll_w:
7960 case Intrinsic::loongarch_lsx_vsll_d:
7961 case Intrinsic::loongarch_lasx_xvsll_b:
7962 case Intrinsic::loongarch_lasx_xvsll_h:
7963 case Intrinsic::loongarch_lasx_xvsll_w:
7964 case Intrinsic::loongarch_lasx_xvsll_d:
7965 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7966 truncateVecElts(N, DAG));
7967 case Intrinsic::loongarch_lsx_vslli_b:
7968 case Intrinsic::loongarch_lasx_xvslli_b:
7969 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7970 lowerVectorSplatImm<3>(N, 2, DAG));
7971 case Intrinsic::loongarch_lsx_vslli_h:
7972 case Intrinsic::loongarch_lasx_xvslli_h:
7973 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7974 lowerVectorSplatImm<4>(N, 2, DAG));
7975 case Intrinsic::loongarch_lsx_vslli_w:
7976 case Intrinsic::loongarch_lasx_xvslli_w:
7977 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7978 lowerVectorSplatImm<5>(N, 2, DAG));
7979 case Intrinsic::loongarch_lsx_vslli_d:
7980 case Intrinsic::loongarch_lasx_xvslli_d:
7981 return DAG.getNode(ISD::SHL, DL, N->getValueType(0), N->getOperand(1),
7982 lowerVectorSplatImm<6>(N, 2, DAG));
7983 case Intrinsic::loongarch_lsx_vsrl_b:
7984 case Intrinsic::loongarch_lsx_vsrl_h:
7985 case Intrinsic::loongarch_lsx_vsrl_w:
7986 case Intrinsic::loongarch_lsx_vsrl_d:
7987 case Intrinsic::loongarch_lasx_xvsrl_b:
7988 case Intrinsic::loongarch_lasx_xvsrl_h:
7989 case Intrinsic::loongarch_lasx_xvsrl_w:
7990 case Intrinsic::loongarch_lasx_xvsrl_d:
7991 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7992 truncateVecElts(N, DAG));
7993 case Intrinsic::loongarch_lsx_vsrli_b:
7994 case Intrinsic::loongarch_lasx_xvsrli_b:
7995 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
7996 lowerVectorSplatImm<3>(N, 2, DAG));
7997 case Intrinsic::loongarch_lsx_vsrli_h:
7998 case Intrinsic::loongarch_lasx_xvsrli_h:
7999 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
8000 lowerVectorSplatImm<4>(N, 2, DAG));
8001 case Intrinsic::loongarch_lsx_vsrli_w:
8002 case Intrinsic::loongarch_lasx_xvsrli_w:
8003 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
8004 lowerVectorSplatImm<5>(N, 2, DAG));
8005 case Intrinsic::loongarch_lsx_vsrli_d:
8006 case Intrinsic::loongarch_lasx_xvsrli_d:
8007 return DAG.getNode(ISD::SRL, DL, N->getValueType(0), N->getOperand(1),
8008 lowerVectorSplatImm<6>(N, 2, DAG));
8009 case Intrinsic::loongarch_lsx_vsra_b:
8010 case Intrinsic::loongarch_lsx_vsra_h:
8011 case Intrinsic::loongarch_lsx_vsra_w:
8012 case Intrinsic::loongarch_lsx_vsra_d:
8013 case Intrinsic::loongarch_lasx_xvsra_b:
8014 case Intrinsic::loongarch_lasx_xvsra_h:
8015 case Intrinsic::loongarch_lasx_xvsra_w:
8016 case Intrinsic::loongarch_lasx_xvsra_d:
8017 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
8018 truncateVecElts(N, DAG));
8019 case Intrinsic::loongarch_lsx_vsrai_b:
8020 case Intrinsic::loongarch_lasx_xvsrai_b:
8021 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
8022 lowerVectorSplatImm<3>(N, 2, DAG));
8023 case Intrinsic::loongarch_lsx_vsrai_h:
8024 case Intrinsic::loongarch_lasx_xvsrai_h:
8025 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
8026 lowerVectorSplatImm<4>(N, 2, DAG));
8027 case Intrinsic::loongarch_lsx_vsrai_w:
8028 case Intrinsic::loongarch_lasx_xvsrai_w:
8029 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
8030 lowerVectorSplatImm<5>(N, 2, DAG));
8031 case Intrinsic::loongarch_lsx_vsrai_d:
8032 case Intrinsic::loongarch_lasx_xvsrai_d:
8033 return DAG.getNode(ISD::SRA, DL, N->getValueType(0), N->getOperand(1),
8034 lowerVectorSplatImm<6>(N, 2, DAG));
8035 case Intrinsic::loongarch_lsx_vclz_b:
8036 case Intrinsic::loongarch_lsx_vclz_h:
8037 case Intrinsic::loongarch_lsx_vclz_w:
8038 case Intrinsic::loongarch_lsx_vclz_d:
8039 case Intrinsic::loongarch_lasx_xvclz_b:
8040 case Intrinsic::loongarch_lasx_xvclz_h:
8041 case Intrinsic::loongarch_lasx_xvclz_w:
8042 case Intrinsic::loongarch_lasx_xvclz_d:
8043 return DAG.getNode(ISD::CTLZ, DL, N->getValueType(0), N->getOperand(1));
8044 case Intrinsic::loongarch_lsx_vpcnt_b:
8045 case Intrinsic::loongarch_lsx_vpcnt_h:
8046 case Intrinsic::loongarch_lsx_vpcnt_w:
8047 case Intrinsic::loongarch_lsx_vpcnt_d:
8048 case Intrinsic::loongarch_lasx_xvpcnt_b:
8049 case Intrinsic::loongarch_lasx_xvpcnt_h:
8050 case Intrinsic::loongarch_lasx_xvpcnt_w:
8051 case Intrinsic::loongarch_lasx_xvpcnt_d:
8052 return DAG.getNode(ISD::CTPOP, DL, N->getValueType(0), N->getOperand(1));
8053 case Intrinsic::loongarch_lsx_vbitclr_b:
8054 case Intrinsic::loongarch_lsx_vbitclr_h:
8055 case Intrinsic::loongarch_lsx_vbitclr_w:
8056 case Intrinsic::loongarch_lsx_vbitclr_d:
8057 case Intrinsic::loongarch_lasx_xvbitclr_b:
8058 case Intrinsic::loongarch_lasx_xvbitclr_h:
8059 case Intrinsic::loongarch_lasx_xvbitclr_w:
8060 case Intrinsic::loongarch_lasx_xvbitclr_d:
8061 return lowerVectorBitClear(N, DAG);
8062 case Intrinsic::loongarch_lsx_vbitclri_b:
8063 case Intrinsic::loongarch_lasx_xvbitclri_b:
8064 return lowerVectorBitClearImm<3>(N, DAG);
8065 case Intrinsic::loongarch_lsx_vbitclri_h:
8066 case Intrinsic::loongarch_lasx_xvbitclri_h:
8067 return lowerVectorBitClearImm<4>(N, DAG);
8068 case Intrinsic::loongarch_lsx_vbitclri_w:
8069 case Intrinsic::loongarch_lasx_xvbitclri_w:
8070 return lowerVectorBitClearImm<5>(N, DAG);
8071 case Intrinsic::loongarch_lsx_vbitclri_d:
8072 case Intrinsic::loongarch_lasx_xvbitclri_d:
8073 return lowerVectorBitClearImm<6>(N, DAG);
8074 case Intrinsic::loongarch_lsx_vbitset_b:
8075 case Intrinsic::loongarch_lsx_vbitset_h:
8076 case Intrinsic::loongarch_lsx_vbitset_w:
8077 case Intrinsic::loongarch_lsx_vbitset_d:
8078 case Intrinsic::loongarch_lasx_xvbitset_b:
8079 case Intrinsic::loongarch_lasx_xvbitset_h:
8080 case Intrinsic::loongarch_lasx_xvbitset_w:
8081 case Intrinsic::loongarch_lasx_xvbitset_d: {
8082 EVT VecTy = N->getValueType(0);
8083 SDValue One = DAG.getConstant(1, DL, VecTy);
8084 return DAG.getNode(
8085 ISD::OR, DL, VecTy, N->getOperand(1),
8086 DAG.getNode(ISD::SHL, DL, VecTy, One, truncateVecElts(N, DAG)));
8087 }
8088 case Intrinsic::loongarch_lsx_vbitseti_b:
8089 case Intrinsic::loongarch_lasx_xvbitseti_b:
8090 return lowerVectorBitSetImm<3>(N, DAG);
8091 case Intrinsic::loongarch_lsx_vbitseti_h:
8092 case Intrinsic::loongarch_lasx_xvbitseti_h:
8093 return lowerVectorBitSetImm<4>(N, DAG);
8094 case Intrinsic::loongarch_lsx_vbitseti_w:
8095 case Intrinsic::loongarch_lasx_xvbitseti_w:
8096 return lowerVectorBitSetImm<5>(N, DAG);
8097 case Intrinsic::loongarch_lsx_vbitseti_d:
8098 case Intrinsic::loongarch_lasx_xvbitseti_d:
8099 return lowerVectorBitSetImm<6>(N, DAG);
8100 case Intrinsic::loongarch_lsx_vbitrev_b:
8101 case Intrinsic::loongarch_lsx_vbitrev_h:
8102 case Intrinsic::loongarch_lsx_vbitrev_w:
8103 case Intrinsic::loongarch_lsx_vbitrev_d:
8104 case Intrinsic::loongarch_lasx_xvbitrev_b:
8105 case Intrinsic::loongarch_lasx_xvbitrev_h:
8106 case Intrinsic::loongarch_lasx_xvbitrev_w:
8107 case Intrinsic::loongarch_lasx_xvbitrev_d: {
8108 EVT VecTy = N->getValueType(0);
8109 SDValue One = DAG.getConstant(1, DL, VecTy);
8110 return DAG.getNode(
8111 ISD::XOR, DL, VecTy, N->getOperand(1),
8112 DAG.getNode(ISD::SHL, DL, VecTy, One, truncateVecElts(N, DAG)));
8113 }
8114 case Intrinsic::loongarch_lsx_vbitrevi_b:
8115 case Intrinsic::loongarch_lasx_xvbitrevi_b:
8116 return lowerVectorBitRevImm<3>(N, DAG);
8117 case Intrinsic::loongarch_lsx_vbitrevi_h:
8118 case Intrinsic::loongarch_lasx_xvbitrevi_h:
8119 return lowerVectorBitRevImm<4>(N, DAG);
8120 case Intrinsic::loongarch_lsx_vbitrevi_w:
8121 case Intrinsic::loongarch_lasx_xvbitrevi_w:
8122 return lowerVectorBitRevImm<5>(N, DAG);
8123 case Intrinsic::loongarch_lsx_vbitrevi_d:
8124 case Intrinsic::loongarch_lasx_xvbitrevi_d:
8125 return lowerVectorBitRevImm<6>(N, DAG);
8126 case Intrinsic::loongarch_lsx_vfadd_s:
8127 case Intrinsic::loongarch_lsx_vfadd_d:
8128 case Intrinsic::loongarch_lasx_xvfadd_s:
8129 case Intrinsic::loongarch_lasx_xvfadd_d:
8130 return DAG.getNode(ISD::FADD, DL, N->getValueType(0), N->getOperand(1),
8131 N->getOperand(2));
8132 case Intrinsic::loongarch_lsx_vfsub_s:
8133 case Intrinsic::loongarch_lsx_vfsub_d:
8134 case Intrinsic::loongarch_lasx_xvfsub_s:
8135 case Intrinsic::loongarch_lasx_xvfsub_d:
8136 return DAG.getNode(ISD::FSUB, DL, N->getValueType(0), N->getOperand(1),
8137 N->getOperand(2));
8138 case Intrinsic::loongarch_lsx_vfmul_s:
8139 case Intrinsic::loongarch_lsx_vfmul_d:
8140 case Intrinsic::loongarch_lasx_xvfmul_s:
8141 case Intrinsic::loongarch_lasx_xvfmul_d:
8142 return DAG.getNode(ISD::FMUL, DL, N->getValueType(0), N->getOperand(1),
8143 N->getOperand(2));
8144 case Intrinsic::loongarch_lsx_vfdiv_s:
8145 case Intrinsic::loongarch_lsx_vfdiv_d:
8146 case Intrinsic::loongarch_lasx_xvfdiv_s:
8147 case Intrinsic::loongarch_lasx_xvfdiv_d:
8148 return DAG.getNode(ISD::FDIV, DL, N->getValueType(0), N->getOperand(1),
8149 N->getOperand(2));
8150 case Intrinsic::loongarch_lsx_vfmadd_s:
8151 case Intrinsic::loongarch_lsx_vfmadd_d:
8152 case Intrinsic::loongarch_lasx_xvfmadd_s:
8153 case Intrinsic::loongarch_lasx_xvfmadd_d:
8154 return DAG.getNode(ISD::FMA, DL, N->getValueType(0), N->getOperand(1),
8155 N->getOperand(2), N->getOperand(3));
8156 case Intrinsic::loongarch_lsx_vinsgr2vr_b:
8157 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
8158 N->getOperand(1), N->getOperand(2),
8159 legalizeIntrinsicImmArg<4>(N, 3, DAG, Subtarget));
8160 case Intrinsic::loongarch_lsx_vinsgr2vr_h:
8161 case Intrinsic::loongarch_lasx_xvinsgr2vr_w:
8162 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
8163 N->getOperand(1), N->getOperand(2),
8164 legalizeIntrinsicImmArg<3>(N, 3, DAG, Subtarget));
8165 case Intrinsic::loongarch_lsx_vinsgr2vr_w:
8166 case Intrinsic::loongarch_lasx_xvinsgr2vr_d:
8167 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
8168 N->getOperand(1), N->getOperand(2),
8169 legalizeIntrinsicImmArg<2>(N, 3, DAG, Subtarget));
8170 case Intrinsic::loongarch_lsx_vinsgr2vr_d:
8171 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), N->getValueType(0),
8172 N->getOperand(1), N->getOperand(2),
8173 legalizeIntrinsicImmArg<1>(N, 3, DAG, Subtarget));
8174 case Intrinsic::loongarch_lsx_vreplgr2vr_b:
8175 case Intrinsic::loongarch_lsx_vreplgr2vr_h:
8176 case Intrinsic::loongarch_lsx_vreplgr2vr_w:
8177 case Intrinsic::loongarch_lsx_vreplgr2vr_d:
8178 case Intrinsic::loongarch_lasx_xvreplgr2vr_b:
8179 case Intrinsic::loongarch_lasx_xvreplgr2vr_h:
8180 case Intrinsic::loongarch_lasx_xvreplgr2vr_w:
8181 case Intrinsic::loongarch_lasx_xvreplgr2vr_d:
8182 return DAG.getNode(LoongArchISD::VREPLGR2VR, DL, N->getValueType(0),
8183 DAG.getNode(ISD::ANY_EXTEND, DL, Subtarget.getGRLenVT(),
8184 N->getOperand(1)));
8185 case Intrinsic::loongarch_lsx_vreplve_b:
8186 case Intrinsic::loongarch_lsx_vreplve_h:
8187 case Intrinsic::loongarch_lsx_vreplve_w:
8188 case Intrinsic::loongarch_lsx_vreplve_d:
8189 case Intrinsic::loongarch_lasx_xvreplve_b:
8190 case Intrinsic::loongarch_lasx_xvreplve_h:
8191 case Intrinsic::loongarch_lasx_xvreplve_w:
8192 case Intrinsic::loongarch_lasx_xvreplve_d:
8193 return DAG.getNode(LoongArchISD::VREPLVE, DL, N->getValueType(0),
8194 N->getOperand(1),
8195 DAG.getNode(ISD::ANY_EXTEND, DL, Subtarget.getGRLenVT(),
8196 N->getOperand(2)));
8197 case Intrinsic::loongarch_lsx_vpickve2gr_b:
8198 if (!Subtarget.is64Bit())
8199 return lowerVectorPickVE2GR<4>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8200 break;
8201 case Intrinsic::loongarch_lsx_vpickve2gr_h:
8202 case Intrinsic::loongarch_lasx_xvpickve2gr_w:
8203 if (!Subtarget.is64Bit())
8204 return lowerVectorPickVE2GR<3>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8205 break;
8206 case Intrinsic::loongarch_lsx_vpickve2gr_w:
8207 if (!Subtarget.is64Bit())
8208 return lowerVectorPickVE2GR<2>(N, DAG, LoongArchISD::VPICK_SEXT_ELT);
8209 break;
8210 case Intrinsic::loongarch_lsx_vpickve2gr_bu:
8211 if (!Subtarget.is64Bit())
8212 return lowerVectorPickVE2GR<4>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8213 break;
8214 case Intrinsic::loongarch_lsx_vpickve2gr_hu:
8215 case Intrinsic::loongarch_lasx_xvpickve2gr_wu:
8216 if (!Subtarget.is64Bit())
8217 return lowerVectorPickVE2GR<3>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8218 break;
8219 case Intrinsic::loongarch_lsx_vpickve2gr_wu:
8220 if (!Subtarget.is64Bit())
8221 return lowerVectorPickVE2GR<2>(N, DAG, LoongArchISD::VPICK_ZEXT_ELT);
8222 break;
8223 case Intrinsic::loongarch_lsx_bz_b:
8224 case Intrinsic::loongarch_lsx_bz_h:
8225 case Intrinsic::loongarch_lsx_bz_w:
8226 case Intrinsic::loongarch_lsx_bz_d:
8227 case Intrinsic::loongarch_lasx_xbz_b:
8228 case Intrinsic::loongarch_lasx_xbz_h:
8229 case Intrinsic::loongarch_lasx_xbz_w:
8230 case Intrinsic::loongarch_lasx_xbz_d:
8231 if (!Subtarget.is64Bit())
8232 return DAG.getNode(LoongArchISD::VALL_ZERO, DL, N->getValueType(0),
8233 N->getOperand(1));
8234 break;
8235 case Intrinsic::loongarch_lsx_bz_v:
8236 case Intrinsic::loongarch_lasx_xbz_v:
8237 if (!Subtarget.is64Bit())
8238 return DAG.getNode(LoongArchISD::VANY_ZERO, DL, N->getValueType(0),
8239 N->getOperand(1));
8240 break;
8241 case Intrinsic::loongarch_lsx_bnz_b:
8242 case Intrinsic::loongarch_lsx_bnz_h:
8243 case Intrinsic::loongarch_lsx_bnz_w:
8244 case Intrinsic::loongarch_lsx_bnz_d:
8245 case Intrinsic::loongarch_lasx_xbnz_b:
8246 case Intrinsic::loongarch_lasx_xbnz_h:
8247 case Intrinsic::loongarch_lasx_xbnz_w:
8248 case Intrinsic::loongarch_lasx_xbnz_d:
8249 if (!Subtarget.is64Bit())
8250 return DAG.getNode(LoongArchISD::VALL_NONZERO, DL, N->getValueType(0),
8251 N->getOperand(1));
8252 break;
8253 case Intrinsic::loongarch_lsx_bnz_v:
8254 case Intrinsic::loongarch_lasx_xbnz_v:
8255 if (!Subtarget.is64Bit())
8256 return DAG.getNode(LoongArchISD::VANY_NONZERO, DL, N->getValueType(0),
8257 N->getOperand(1));
8258 break;
8259 case Intrinsic::loongarch_lasx_concat_128_s:
8260 case Intrinsic::loongarch_lasx_concat_128_d:
8261 case Intrinsic::loongarch_lasx_concat_128:
8262 return DAG.getNode(ISD::CONCAT_VECTORS, DL, N->getValueType(0),
8263 N->getOperand(1), N->getOperand(2));
8264 }
8265 return SDValue();
8266}
8267
8270 const LoongArchSubtarget &Subtarget) {
8271 // If the input to MOVGR2FR_W_LA64 is just MOVFR2GR_S_LA64 the the
8272 // conversion is unnecessary and can be replaced with the
8273 // MOVFR2GR_S_LA64 operand.
8274 SDValue Op0 = N->getOperand(0);
8275 if (Op0.getOpcode() == LoongArchISD::MOVFR2GR_S_LA64)
8276 return Op0.getOperand(0);
8277 return SDValue();
8278}
8279
8282 const LoongArchSubtarget &Subtarget) {
8283 // If the input to MOVFR2GR_S_LA64 is just MOVGR2FR_W_LA64 then the
8284 // conversion is unnecessary and can be replaced with the MOVGR2FR_W_LA64
8285 // operand.
8286 SDValue Op0 = N->getOperand(0);
8287 if (Op0->getOpcode() == LoongArchISD::MOVGR2FR_W_LA64) {
8288 assert(Op0.getOperand(0).getValueType() == N->getSimpleValueType(0) &&
8289 "Unexpected value type!");
8290 return Op0.getOperand(0);
8291 }
8292 return SDValue();
8293}
8294
8295static SDValue
8298 MVT VT = N->getSimpleValueType(0);
8299 unsigned NumBits = VT.getScalarSizeInBits();
8300
8301 // Simplify the inputs.
8302 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
8303 APInt DemandedMask(APInt::getAllOnes(NumBits));
8304 if (TLI.SimplifyDemandedBits(SDValue(N, 0), DemandedMask, DCI))
8305 return SDValue(N, 0);
8306
8307 return SDValue();
8308}
8309
8310static SDValue
8313 const LoongArchSubtarget &Subtarget) {
8314 SDValue Op0 = N->getOperand(0);
8315 SDLoc DL(N);
8316
8317 // If the input to SplitPairF64 is just BuildPairF64 then the operation is
8318 // redundant. Instead, use BuildPairF64's operands directly.
8319 if (Op0->getOpcode() == LoongArchISD::BUILD_PAIR_F64)
8320 return DCI.CombineTo(N, Op0.getOperand(0), Op0.getOperand(1));
8321
8322 if (Op0->isUndef()) {
8323 SDValue Lo = DAG.getUNDEF(MVT::i32);
8324 SDValue Hi = DAG.getUNDEF(MVT::i32);
8325 return DCI.CombineTo(N, Lo, Hi);
8326 }
8327
8328 // It's cheaper to materialise two 32-bit integers than to load a double
8329 // from the constant pool and transfer it to integer registers through the
8330 // stack.
8332 APInt V = C->getValueAPF().bitcastToAPInt();
8333 SDValue Lo = DAG.getConstant(V.trunc(32), DL, MVT::i32);
8334 SDValue Hi = DAG.getConstant(V.lshr(32).trunc(32), DL, MVT::i32);
8335 return DCI.CombineTo(N, Lo, Hi);
8336 }
8337
8338 return SDValue();
8339}
8340
8341/// Do target-specific dag combines on LoongArchISD::VANDN nodes.
8344 const LoongArchSubtarget &Subtarget) {
8345 SDValue N0 = N->getOperand(0);
8346 SDValue N1 = N->getOperand(1);
8347 MVT VT = N->getSimpleValueType(0);
8348 SDLoc DL(N);
8349
8350 // VANDN(undef, x) -> 0
8351 // VANDN(x, undef) -> 0
8352 if (N0.isUndef() || N1.isUndef())
8353 return DAG.getConstant(0, DL, VT);
8354
8355 // VANDN(0, x) -> x
8357 return N1;
8358
8359 // VANDN(x, 0) -> 0
8361 return DAG.getConstant(0, DL, VT);
8362
8363 // VANDN(x, -1) -> NOT(x) -> XOR(x, -1)
8365 return DAG.getNOT(DL, N0, VT);
8366
8367 // Turn VANDN back to AND if input is inverted.
8368 if (SDValue Not = isNOT(N0, DAG))
8369 return DAG.getNode(ISD::AND, DL, VT, DAG.getBitcast(VT, Not), N1);
8370
8371 // Folds for better commutativity:
8372 if (N1->hasOneUse()) {
8373 // VANDN(x,NOT(y)) -> AND(NOT(x),NOT(y)) -> NOT(OR(X,Y)).
8374 if (SDValue Not = isNOT(N1, DAG))
8375 return DAG.getNOT(
8376 DL, DAG.getNode(ISD::OR, DL, VT, N0, DAG.getBitcast(VT, Not)), VT);
8377
8378 // VANDN(x, SplatVector(Imm)) -> AND(NOT(x), NOT(SplatVector(~Imm)))
8379 // -> NOT(OR(x, SplatVector(-Imm))
8380 // Combination is performed only when VT is v16i8/v32i8, using `vnori.b` to
8381 // gain benefits.
8382 if (!DCI.isBeforeLegalizeOps() && (VT == MVT::v16i8 || VT == MVT::v32i8) &&
8383 N1.getOpcode() == ISD::BUILD_VECTOR) {
8384 if (SDValue SplatValue =
8385 cast<BuildVectorSDNode>(N1.getNode())->getSplatValue()) {
8386 if (!N1->isOnlyUserOf(SplatValue.getNode()))
8387 return SDValue();
8388
8389 if (auto *C = dyn_cast<ConstantSDNode>(SplatValue)) {
8390 uint8_t NCVal = static_cast<uint8_t>(~(C->getSExtValue()));
8391 SDValue Not =
8392 DAG.getSplat(VT, DL, DAG.getTargetConstant(NCVal, DL, MVT::i8));
8393 return DAG.getNOT(
8394 DL, DAG.getNode(ISD::OR, DL, VT, N0, DAG.getBitcast(VT, Not)),
8395 VT);
8396 }
8397 }
8398 }
8399 }
8400
8401 return SDValue();
8402}
8403
8404static SDValue ExtendSrcToDst(SDNode *N, SelectionDAG &DAG, unsigned ExtendOp) {
8405 SDLoc DL(N);
8406 EVT VT = N->getValueType(0);
8407 SDValue Src = N->getOperand(0);
8408 EVT SrcVT = Src.getValueType();
8409
8410 unsigned DstElts = VT.getVectorNumElements();
8411 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8412 unsigned DstEltBits = VT.getScalarSizeInBits();
8413
8414 if (SrcEltBits >= DstEltBits)
8415 return SDValue();
8416
8417 MVT WidenEltVT = MVT::getIntegerVT(DstEltBits);
8418 EVT WidenSrcVT = EVT::getVectorVT(*DAG.getContext(), WidenEltVT, DstElts);
8419
8420 SDValue Extend = DAG.getNode(ExtendOp, DL, WidenSrcVT, Src);
8421 return DAG.getNode(N->getOpcode(), DL, VT, Extend);
8422}
8423
8424// Merge two 64 to 32 convert instructions into one,
8425// e.g.
8426// vffint.s.l $vr0, $vr1, $vr2
8427// will convert 4 si64 into 4 float at once.
8428// or
8429// vftintrz.w.d $vr0, $vr1, $vr2
8430// which will convert 4 double into 4 si32 at once.
8431// also deal with their 256-bits LASX version.
8432static SDValue MergeBlocksConvert(SDNode *N, SelectionDAG &DAG, unsigned Opcode,
8433 unsigned BlockBits) {
8434 SDLoc DL(N);
8435 MVT DstVT = N->getSimpleValueType(0);
8436 SDValue Src = N->getOperand(0);
8437 MVT SrcVT = Src.getSimpleValueType();
8438 unsigned SrcBits = SrcVT.getSizeInBits();
8439
8441 unsigned BlockNumElts = BlockBits / SrcVT.getScalarSizeInBits();
8442 MVT BlockVT = MVT::getVectorVT(SrcVT.getScalarType(), BlockNumElts);
8443 if (Src.getOpcode() == ISD::CONCAT_VECTORS &&
8444 Src.getOperand(0).getValueType() == BlockVT) {
8445 for (unsigned i = 0; i < Src.getNumOperands(); ++i)
8446 Blocks.push_back(Src.getOperand(i));
8447 } else if (SrcBits > BlockBits) {
8448 // Wider than one register: extract each BlockBits-wide sub-vector.
8449 for (unsigned i = 0; i < SrcBits / BlockBits; ++i)
8450 Blocks.push_back(
8451 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, BlockVT, Src,
8452 DAG.getVectorIdxConstant(i * BlockNumElts, DL)));
8453 } else {
8454 BlockBits = SrcBits;
8455 Blocks.push_back(Src);
8456 }
8457
8458 MVT NativeVecVT = MVT::getVectorVT(DstVT.getScalarType(),
8459 BlockBits / DstVT.getScalarSizeInBits());
8461 for (unsigned i = 0; i < Blocks.size(); i += 2) {
8462 SDValue Lo = Blocks[i];
8463 SDValue Hi = Blocks.size() > 1 ? Blocks[i + 1] : Lo;
8464 SDValue Res = DAG.getNode(Opcode, DL, NativeVecVT, Hi, Lo);
8465
8466 if (BlockBits == 256) {
8467 SDValue Undef = DAG.getUNDEF(NativeVecVT);
8468 SmallVector<int, 8> Mask = {0, 1, 4, 5, 2, 3, 6, 7};
8469 Res = DAG.getVectorShuffle(NativeVecVT, DL, Res, Undef, Mask);
8470 Res = DAG.getBitcast(NativeVecVT, Res);
8471 }
8472
8473 Parts.push_back(Res);
8474 }
8475
8476 if (Blocks.size() == 1)
8477 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, DstVT, Parts[0],
8478 DAG.getVectorIdxConstant(0, DL));
8479 return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, Parts);
8480}
8481
8484 const LoongArchSubtarget &Subtarget) {
8485 SDLoc DL(N);
8486 EVT VT = N->getValueType(0);
8487 SDValue Src = N->getOperand(0);
8488 EVT SrcVT = Src.getValueType();
8489
8490 if (VT.isVector()) {
8491 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8492 unsigned DstEltBits = VT.getScalarSizeInBits();
8493 unsigned NumElts = VT.getVectorNumElements();
8494 unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
8495
8496 // Sign-extend src to avoid scalarization.
8497 if (SrcEltBits <= DstEltBits)
8498 return ExtendSrcToDst(N, DAG, ISD::SIGN_EXTEND);
8499
8500 if (SrcEltBits != 64 || DstEltBits != 32 || !isPowerOf2_32(NumElts))
8501 return SDValue();
8502
8503 if (!SrcVT.isSimple() || !VT.isSimple())
8504 return SDValue();
8505
8506 // Combine [x]vffint.s.l for vector si64 to float conversion.
8507 return MergeBlocksConvert(N, DAG, LoongArchISD::VFFINT, BlockBits);
8508 }
8509
8510 if (VT != MVT::f32 && VT != MVT::f64)
8511 return SDValue();
8512 if (VT == MVT::f32 && !Subtarget.hasBasicF())
8513 return SDValue();
8514 if (VT == MVT::f64 && !Subtarget.hasBasicD())
8515 return SDValue();
8516
8517 // Only optimize when the source and destination types have the same width.
8518 if (VT.getSizeInBits() != N->getOperand(0).getValueSizeInBits())
8519 return SDValue();
8520
8521 // If the result of an integer load is only used by an integer-to-float
8522 // conversion, use a fp load instead. This eliminates an integer-to-float-move
8523 // (movgr2fr) instruction.
8524 if (ISD::isNormalLoad(Src.getNode()) && Src.hasOneUse() &&
8525 // Do not change the width of a volatile load. This condition check is
8526 // inspired by AArch64.
8527 !cast<LoadSDNode>(Src)->isVolatile()) {
8528 LoadSDNode *LN0 = cast<LoadSDNode>(Src);
8529 SDValue Load = DAG.getLoad(VT, DL, LN0->getChain(), LN0->getBasePtr(),
8530 LN0->getPointerInfo(), LN0->getAlign(),
8531 LN0->getMemOperand()->getFlags());
8532
8533 // Make sure successors of the original load stay after it by updating them
8534 // to use the new Chain.
8535 DAG.ReplaceAllUsesOfValueWith(SDValue(LN0, 1), Load.getValue(1));
8536 return DAG.getNode(LoongArchISD::SITOF, SDLoc(N), VT, Load);
8537 }
8538
8539 return SDValue();
8540}
8541
8544 const LoongArchSubtarget &Subtarget) {
8545 SDLoc DL(N);
8546 EVT VT = N->getValueType(0);
8547
8548 // Zero-extend src to avoid scalarization.
8549 if (VT.isVector())
8550 return ExtendSrcToDst(N, DAG, ISD::ZERO_EXTEND);
8551
8552 return SDValue();
8553}
8554
8555// Using [X]VFTINTRZ_W_D for double to signed 32-bit integer conversion.
8556// For example:
8557// v4i32 = fp_to_sint (concat_vectors v2f64, v2f64)
8558// Can be combined into:
8559// v4i32 = VFTINTRZ_W_D v2f64. v2f64
8562 const LoongArchSubtarget &Subtarget) {
8563 if (!Subtarget.hasExtLSX())
8564 return SDValue();
8565
8566 SDLoc DL(N);
8567 EVT DstVT = N->getValueType(0);
8568 SDValue Src = N->getOperand(0);
8569 EVT SrcVT = Src.getValueType();
8570 bool IsSigned = N->getOpcode() == ISD::FP_TO_SINT;
8571
8572 if (!DstVT.isVector() || !DstVT.isSimple() || !SrcVT.isSimple())
8573 return SDValue();
8574
8575 unsigned SrcEltBits = SrcVT.getScalarSizeInBits();
8576 unsigned SrcBits = SrcVT.getSizeInBits();
8577 unsigned DstEltBits = DstVT.getScalarSizeInBits();
8578 unsigned NumElts = DstVT.getVectorNumElements();
8579 unsigned BlockBits = Subtarget.hasExtLASX() ? 256 : 128;
8580
8581 if (!isPowerOf2_32(NumElts) || !isPowerOf2_32(DstEltBits))
8582 return SDValue();
8583
8584 if (SrcBits % BlockBits != 0 && SrcBits != 128)
8585 return SDValue();
8586
8587 if (DstEltBits < 32) {
8588 MVT PromoteVT = MVT::getVectorVT(MVT::getIntegerVT(32), NumElts);
8589 SDValue Conv = DAG.getNode(N->getOpcode(), DL, PromoteVT, Src);
8590 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Conv);
8591 }
8592
8593 if (SrcEltBits != 64 || DstEltBits != 32)
8594 return SDValue();
8595
8596 if (!IsSigned) {
8597 // LASX already has pattern for double convert to uint32.
8598 if (Subtarget.hasExtLASX())
8599 return SDValue();
8600 MVT TmpVT = MVT::getVectorVT(MVT::i64, NumElts);
8601 SDValue Tmp = DAG.getNode(ISD::FP_TO_SINT, DL, TmpVT, Src);
8602 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, Tmp);
8603 }
8604
8605 return MergeBlocksConvert(N, DAG, LoongArchISD::VFTINTRZ, BlockBits);
8606}
8607
8608// Try to widen AND, OR and XOR nodes to VT in order to remove casts around
8609// logical operations, like in the example below.
8610// or (and (truncate x, truncate y)),
8611// (xor (truncate z, build_vector (constants)))
8612// Given a target type \p VT, we generate
8613// or (and x, y), (xor z, zext(build_vector (constants)))
8614// given x, y and z are of type \p VT. We can do so, if operands are either
8615// truncates from VT types, the second operand is a vector of constants, can
8616// be recursively promoted or is an existing extension we can extend further.
8618 SelectionDAG &DAG,
8619 const LoongArchSubtarget &Subtarget,
8620 unsigned Depth) {
8621 // Limit recursion to avoid excessive compile times.
8623 return SDValue();
8624
8625 if (!ISD::isBitwiseLogicOp(N.getOpcode()))
8626 return SDValue();
8627
8628 SDValue N0 = N.getOperand(0);
8629 SDValue N1 = N.getOperand(1);
8630
8631 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
8632 if (!TLI.isOperationLegalOrPromote(N.getOpcode(), VT))
8633 return SDValue();
8634
8635 if (SDValue NN0 =
8636 PromoteMaskArithmetic(N0, DL, VT, DAG, Subtarget, Depth + 1))
8637 N0 = NN0;
8638 else {
8639 // The left side has to be a 'trunc'.
8640 bool LHSTrunc = N0.getOpcode() == ISD::TRUNCATE &&
8641 N0.getOperand(0).getValueType() == VT;
8642 if (LHSTrunc)
8643 N0 = N0.getOperand(0);
8644 else
8645 return SDValue();
8646 }
8647
8648 if (SDValue NN1 =
8649 PromoteMaskArithmetic(N1, DL, VT, DAG, Subtarget, Depth + 1))
8650 N1 = NN1;
8651 else {
8652 // The right side has to be a 'trunc', a (foldable) constant or an
8653 // existing extension we can extend further.
8654 bool RHSTrunc = N1.getOpcode() == ISD::TRUNCATE &&
8655 N1.getOperand(0).getValueType() == VT;
8656 if (RHSTrunc)
8657 N1 = N1.getOperand(0);
8658 else if (ISD::isExtVecInRegOpcode(N1.getOpcode()) && VT.is256BitVector() &&
8659 Subtarget.hasExtLASX() && N1.hasOneUse())
8660 N1 = DAG.getNode(N1.getOpcode(), DL, VT, N1.getOperand(0));
8661 // On 32-bit platform, i64 is an illegal integer scalar type, and
8662 // FoldConstantArithmetic will fail for v4i64. This may be optimized in the
8663 // future.
8664 else if (SDValue Cst =
8666 N1 = Cst;
8667 else
8668 return SDValue();
8669 }
8670
8671 return DAG.getNode(N.getOpcode(), DL, VT, N0, N1);
8672}
8673
8674// On LASX the type v4i1/v8i1/v16i1 may be legalized to v4i32/v8i16/v16i8, which
8675// is LSX-sized register. In most cases we actually compare or select LASX-sized
8676// registers and mixing the two types creates horrible code. This method
8677// optimizes some of the transition sequences.
8679 SelectionDAG &DAG,
8680 const LoongArchSubtarget &Subtarget) {
8681 EVT VT = N.getValueType();
8682 assert(VT.isVector() && "Expected vector type");
8683 assert((N.getOpcode() == ISD::ANY_EXTEND ||
8684 N.getOpcode() == ISD::ZERO_EXTEND ||
8685 N.getOpcode() == ISD::SIGN_EXTEND) &&
8686 "Invalid Node");
8687
8688 if (!Subtarget.hasExtLASX() || !VT.is256BitVector())
8689 return SDValue();
8690
8691 SDValue Narrow = N.getOperand(0);
8692 EVT NarrowVT = Narrow.getValueType();
8693
8694 // Generate the wide operation.
8695 SDValue Op = PromoteMaskArithmetic(Narrow, DL, VT, DAG, Subtarget, 0);
8696 if (!Op)
8697 return SDValue();
8698 switch (N.getOpcode()) {
8699 default:
8700 llvm_unreachable("Unexpected opcode");
8701 case ISD::ANY_EXTEND:
8702 return Op;
8703 case ISD::ZERO_EXTEND:
8704 return DAG.getZeroExtendInReg(Op, DL, NarrowVT);
8705 case ISD::SIGN_EXTEND:
8706 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, Op,
8707 DAG.getValueType(NarrowVT));
8708 }
8709}
8710
8713 const LoongArchSubtarget &Subtarget) {
8714 EVT VT = N->getValueType(0);
8715 SDLoc DL(N);
8716
8717 if (VT.isVector()) {
8718 if (SDValue R = PromoteMaskArithmetic(SDValue(N, 0), DL, DAG, Subtarget))
8719 return R;
8720
8721 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT) ||
8722 N->getValueSizeInBits(0) != N->getOperand(0).getValueSizeInBits() * 2)
8723 return SDValue();
8724
8725 if (SDValue R = matchHalfOf128BitLanes(N->getOperand(0), /*isLow=*/false)) {
8726 if (N->getOpcode() == ISD::SIGN_EXTEND)
8727 return DAG.getNode(LoongArchISD::VEXTH, DL, VT, R);
8728 if (N->getOpcode() == ISD::ZERO_EXTEND)
8729 return DAG.getNode(LoongArchISD::VEXTH_U, DL, VT, R);
8730 }
8731 }
8732
8733 return SDValue();
8734}
8735
8736static SDValue
8739 const LoongArchSubtarget &Subtarget) {
8740 SDLoc DL(N);
8741 EVT VT = N->getValueType(0);
8742
8743 if (VT.isVector() && N->getNumOperands() == 2)
8744 if (SDValue R = combineFP_ROUND(SDValue(N, 0), DL, DAG, Subtarget))
8745 return R;
8746
8747 return SDValue();
8748}
8749
8752 const LoongArchSubtarget &Subtarget) {
8753 if (DCI.isBeforeLegalizeOps())
8754 return SDValue();
8755
8756 EVT VT = N->getValueType(0);
8757 if (!VT.isVector())
8758 return SDValue();
8759
8760 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
8761 return SDValue();
8762
8763 EVT EltVT = VT.getVectorElementType();
8764 if (!EltVT.isInteger())
8765 return SDValue();
8766
8767 SDValue Cond = N->getOperand(0);
8768 SDValue TrueVal = N->getOperand(1);
8769 SDValue FalseVal = N->getOperand(2);
8770
8771 // match:
8772 //
8773 // vselect (setcc shift, 0, seteq),
8774 // x,
8775 // rounded_shift
8776
8777 if (Cond.getOpcode() != ISD::SETCC)
8778 return SDValue();
8779
8780 if (!ISD::isConstantSplatVectorAllZeros(Cond.getOperand(1).getNode()))
8781 return SDValue();
8782
8783 auto *CC = cast<CondCodeSDNode>(Cond.getOperand(2));
8784 if (CC->get() != ISD::SETEQ)
8785 return SDValue();
8786
8787 SDValue Shift = Cond.getOperand(0);
8788
8789 // True branch must be original value:
8790 //
8791 // vselect cond, x, ...
8792
8793 SDValue X = TrueVal;
8794
8795 // Now match rounded shift pattern:
8796 //
8797 // add
8798 // (and
8799 // (srl X, shift-1)
8800 // 1)
8801 // (srl/sra X, shift)
8802
8803 if (FalseVal.getOpcode() != ISD::ADD)
8804 return SDValue();
8805
8806 SDValue Add0 = FalseVal.getOperand(0);
8807 SDValue Add1 = FalseVal.getOperand(1);
8808 SDValue And;
8809 SDValue Shr;
8810
8811 if (Add0.getOpcode() == ISD::AND) {
8812 And = Add0;
8813 Shr = Add1;
8814 } else if (Add1.getOpcode() == ISD::AND) {
8815 And = Add1;
8816 Shr = Add0;
8817 } else {
8818 return SDValue();
8819 }
8820
8821 // match:
8822 //
8823 // srl/sra X, shift
8824
8825 if (Shr.getOpcode() != ISD::SRL && Shr.getOpcode() != ISD::SRA)
8826 return SDValue();
8827
8828 if (Shr.getOperand(0) != X)
8829 return SDValue();
8830
8831 if (Shr.getOperand(1) != Shift)
8832 return SDValue();
8833
8834 // match:
8835 //
8836 // and
8837 // (srl X, shift-1)
8838 // 1
8839
8840 SDValue Srl = And.getOperand(0);
8841 SDValue One = And.getOperand(1);
8842 APInt SplatVal;
8843
8844 if (Srl.getOpcode() != ISD::SRL)
8845 return SDValue();
8846
8847 One = peekThroughBitcasts(One);
8848 if (!isConstantSplatVector(One, SplatVal, EltVT.getSizeInBits()))
8849 return SDValue();
8850
8851 if (SplatVal != 1)
8852 return SDValue();
8853
8854 if (Srl.getOperand(0) != X)
8855 return SDValue();
8856
8857 // match:
8858 //
8859 // shift-1
8860
8861 SDValue ShiftMinus1 = Srl.getOperand(1);
8862
8863 if (ShiftMinus1.getOpcode() != ISD::ADD)
8864 return SDValue();
8865
8866 if (ShiftMinus1.getOperand(0) != Shift)
8867 return SDValue();
8868
8870 return SDValue();
8871
8872 // We matched a rounded right shift pattern and can lower it
8873 // to a single vector rounded shift instruction.
8874
8875 SDLoc DL(N);
8876 return DAG.getNode(Shr.getOpcode() == ISD::SRL ? LoongArchISD::VSRLR
8877 : LoongArchISD::VSRAR,
8878 DL, VT, X, Shift);
8879}
8880
8882 DAGCombinerInfo &DCI) const {
8883 SelectionDAG &DAG = DCI.DAG;
8884 switch (N->getOpcode()) {
8885 default:
8886 break;
8887 case ISD::ADD:
8888 return performADDCombine(N, DAG, DCI, Subtarget);
8889 case ISD::AND:
8890 return performANDCombine(N, DAG, DCI, Subtarget);
8891 case ISD::OR:
8892 return performORCombine(N, DAG, DCI, Subtarget);
8893 case ISD::SETCC:
8894 return performSETCCCombine(N, DAG, DCI, Subtarget);
8895 case ISD::SELECT:
8896 return performSELECTCombine(N, DAG, DCI, Subtarget);
8897 case ISD::SHL:
8898 return performSHLCombine(N, DAG, DCI, Subtarget);
8899 case ISD::SRL:
8900 return performSRLCombine(N, DAG, DCI, Subtarget);
8901 case ISD::SUB:
8902 return performSUBCombine(N, DAG, DCI, Subtarget);
8903 case ISD::BITCAST:
8904 return performBITCASTCombine(N, DAG, DCI, Subtarget);
8905 case ISD::ANY_EXTEND:
8906 case ISD::ZERO_EXTEND:
8907 case ISD::SIGN_EXTEND:
8908 return performEXTENDCombine(N, DAG, DCI, Subtarget);
8909 case ISD::SINT_TO_FP:
8910 return performSINT_TO_FPCombine(N, DAG, DCI, Subtarget);
8911 case ISD::UINT_TO_FP:
8912 return performUINT_TO_FPCombine(N, DAG, DCI, Subtarget);
8913 case ISD::FP_TO_SINT:
8914 case ISD::FP_TO_UINT:
8915 return performFP_TO_INTCombine(N, DAG, DCI, Subtarget);
8916 case LoongArchISD::BITREV_W:
8917 return performBITREV_WCombine(N, DAG, DCI, Subtarget);
8918 case LoongArchISD::BR_CC:
8919 return performBR_CCCombine(N, DAG, DCI, Subtarget);
8920 case LoongArchISD::SELECT_CC:
8921 return performSELECT_CCCombine(N, DAG, DCI, Subtarget);
8923 return performINTRINSIC_WO_CHAINCombine(N, DAG, DCI, Subtarget);
8924 case LoongArchISD::MOVGR2FR_W_LA64:
8925 return performMOVGR2FR_WCombine(N, DAG, DCI, Subtarget);
8926 case LoongArchISD::MOVFR2GR_S_LA64:
8927 return performMOVFR2GR_SCombine(N, DAG, DCI, Subtarget);
8928 case LoongArchISD::CRC_W_B_W:
8929 case LoongArchISD::CRC_W_H_W:
8930 case LoongArchISD::CRCC_W_B_W:
8931 case LoongArchISD::CRCC_W_H_W:
8932 case LoongArchISD::VMSKLTZ:
8933 case LoongArchISD::XVMSKLTZ:
8934 return performDemandedBitsCombine(N, DAG, DCI);
8935 case LoongArchISD::SPLIT_PAIR_F64:
8936 return performSPLIT_PAIR_F64Combine(N, DAG, DCI, Subtarget);
8937 case LoongArchISD::VANDN:
8938 return performVANDNCombine(N, DAG, DCI, Subtarget);
8940 return performCONCAT_VECTORSCombine(N, DAG, DCI, Subtarget);
8941 case ISD::VSELECT:
8942 return performVSELECTCombine(N, DAG, DCI, Subtarget);
8943 case LoongArchISD::VPACKEV:
8944 case LoongArchISD::VPERMI:
8945 if (SDValue Result =
8946 combineFP_ROUND(SDValue(N, 0), SDLoc(N), DAG, Subtarget))
8947 return Result;
8948 }
8949 return SDValue();
8950}
8951
8954 if (!ZeroDivCheck)
8955 return MBB;
8956
8957 // Build instructions:
8958 // MBB:
8959 // div(or mod) $dst, $dividend, $divisor
8960 // bne $divisor, $zero, SinkMBB
8961 // BreakMBB:
8962 // break 7 // BRK_DIVZERO
8963 // SinkMBB:
8964 // fallthrough
8965 const BasicBlock *LLVM_BB = MBB->getBasicBlock();
8966 MachineFunction::iterator It = ++MBB->getIterator();
8967 MachineFunction *MF = MBB->getParent();
8968 auto BreakMBB = MF->CreateMachineBasicBlock(LLVM_BB);
8969 auto SinkMBB = MF->CreateMachineBasicBlock(LLVM_BB);
8970 MF->insert(It, BreakMBB);
8971 MF->insert(It, SinkMBB);
8972
8973 // Transfer the remainder of MBB and its successor edges to SinkMBB.
8974 SinkMBB->splice(SinkMBB->end(), MBB, std::next(MI.getIterator()), MBB->end());
8975 SinkMBB->transferSuccessorsAndUpdatePHIs(MBB);
8976
8977 const TargetInstrInfo &TII = *MF->getSubtarget().getInstrInfo();
8978 DebugLoc DL = MI.getDebugLoc();
8979 MachineOperand &Divisor = MI.getOperand(2);
8980 Register DivisorReg = Divisor.getReg();
8981
8982 // MBB:
8983 BuildMI(MBB, DL, TII.get(LoongArch::BNE))
8984 .addReg(DivisorReg, getKillRegState(Divisor.isKill()))
8985 .addReg(LoongArch::R0)
8986 .addMBB(SinkMBB);
8987 MBB->addSuccessor(BreakMBB);
8988 MBB->addSuccessor(SinkMBB);
8989
8990 // BreakMBB:
8991 // See linux header file arch/loongarch/include/uapi/asm/break.h for the
8992 // definition of BRK_DIVZERO.
8993 BuildMI(BreakMBB, DL, TII.get(LoongArch::BREAK)).addImm(7 /*BRK_DIVZERO*/);
8994 BreakMBB->addSuccessor(SinkMBB);
8995
8996 // Clear Divisor's kill flag.
8997 Divisor.setIsKill(false);
8998
8999 return SinkMBB;
9000}
9001
9002static MachineBasicBlock *
9004 const LoongArchSubtarget &Subtarget) {
9005 unsigned CondOpc;
9006 switch (MI.getOpcode()) {
9007 default:
9008 llvm_unreachable("Unexpected opcode");
9009 case LoongArch::PseudoVBZ:
9010 CondOpc = LoongArch::VSETEQZ_V;
9011 break;
9012 case LoongArch::PseudoVBZ_B:
9013 CondOpc = LoongArch::VSETANYEQZ_B;
9014 break;
9015 case LoongArch::PseudoVBZ_H:
9016 CondOpc = LoongArch::VSETANYEQZ_H;
9017 break;
9018 case LoongArch::PseudoVBZ_W:
9019 CondOpc = LoongArch::VSETANYEQZ_W;
9020 break;
9021 case LoongArch::PseudoVBZ_D:
9022 CondOpc = LoongArch::VSETANYEQZ_D;
9023 break;
9024 case LoongArch::PseudoVBNZ:
9025 CondOpc = LoongArch::VSETNEZ_V;
9026 break;
9027 case LoongArch::PseudoVBNZ_B:
9028 CondOpc = LoongArch::VSETALLNEZ_B;
9029 break;
9030 case LoongArch::PseudoVBNZ_H:
9031 CondOpc = LoongArch::VSETALLNEZ_H;
9032 break;
9033 case LoongArch::PseudoVBNZ_W:
9034 CondOpc = LoongArch::VSETALLNEZ_W;
9035 break;
9036 case LoongArch::PseudoVBNZ_D:
9037 CondOpc = LoongArch::VSETALLNEZ_D;
9038 break;
9039 case LoongArch::PseudoXVBZ:
9040 CondOpc = LoongArch::XVSETEQZ_V;
9041 break;
9042 case LoongArch::PseudoXVBZ_B:
9043 CondOpc = LoongArch::XVSETANYEQZ_B;
9044 break;
9045 case LoongArch::PseudoXVBZ_H:
9046 CondOpc = LoongArch::XVSETANYEQZ_H;
9047 break;
9048 case LoongArch::PseudoXVBZ_W:
9049 CondOpc = LoongArch::XVSETANYEQZ_W;
9050 break;
9051 case LoongArch::PseudoXVBZ_D:
9052 CondOpc = LoongArch::XVSETANYEQZ_D;
9053 break;
9054 case LoongArch::PseudoXVBNZ:
9055 CondOpc = LoongArch::XVSETNEZ_V;
9056 break;
9057 case LoongArch::PseudoXVBNZ_B:
9058 CondOpc = LoongArch::XVSETALLNEZ_B;
9059 break;
9060 case LoongArch::PseudoXVBNZ_H:
9061 CondOpc = LoongArch::XVSETALLNEZ_H;
9062 break;
9063 case LoongArch::PseudoXVBNZ_W:
9064 CondOpc = LoongArch::XVSETALLNEZ_W;
9065 break;
9066 case LoongArch::PseudoXVBNZ_D:
9067 CondOpc = LoongArch::XVSETALLNEZ_D;
9068 break;
9069 }
9070
9071 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
9072 const BasicBlock *LLVM_BB = BB->getBasicBlock();
9073 DebugLoc DL = MI.getDebugLoc();
9076
9077 MachineFunction *F = BB->getParent();
9078 MachineBasicBlock *FalseBB = F->CreateMachineBasicBlock(LLVM_BB);
9079 MachineBasicBlock *TrueBB = F->CreateMachineBasicBlock(LLVM_BB);
9080 MachineBasicBlock *SinkBB = F->CreateMachineBasicBlock(LLVM_BB);
9081
9082 F->insert(It, FalseBB);
9083 F->insert(It, TrueBB);
9084 F->insert(It, SinkBB);
9085
9086 // Transfer the remainder of MBB and its successor edges to Sink.
9087 SinkBB->splice(SinkBB->end(), BB, std::next(MI.getIterator()), BB->end());
9089
9090 // Insert the real instruction to BB.
9091 Register FCC = MRI.createVirtualRegister(&LoongArch::CFRRegClass);
9092 BuildMI(BB, DL, TII->get(CondOpc), FCC).addReg(MI.getOperand(1).getReg());
9093
9094 // Insert branch.
9095 BuildMI(BB, DL, TII->get(LoongArch::BCNEZ)).addReg(FCC).addMBB(TrueBB);
9096 BB->addSuccessor(FalseBB);
9097 BB->addSuccessor(TrueBB);
9098
9099 // FalseBB.
9100 Register RD1 = MRI.createVirtualRegister(&LoongArch::GPRRegClass);
9101 BuildMI(FalseBB, DL, TII->get(LoongArch::ADDI_W), RD1)
9102 .addReg(LoongArch::R0)
9103 .addImm(0);
9104 BuildMI(FalseBB, DL, TII->get(LoongArch::PseudoBR)).addMBB(SinkBB);
9105 FalseBB->addSuccessor(SinkBB);
9106
9107 // TrueBB.
9108 Register RD2 = MRI.createVirtualRegister(&LoongArch::GPRRegClass);
9109 BuildMI(TrueBB, DL, TII->get(LoongArch::ADDI_W), RD2)
9110 .addReg(LoongArch::R0)
9111 .addImm(1);
9112 TrueBB->addSuccessor(SinkBB);
9113
9114 // SinkBB: merge the results.
9115 BuildMI(*SinkBB, SinkBB->begin(), DL, TII->get(LoongArch::PHI),
9116 MI.getOperand(0).getReg())
9117 .addReg(RD1)
9118 .addMBB(FalseBB)
9119 .addReg(RD2)
9120 .addMBB(TrueBB);
9121
9122 // The pseudo instruction is gone now.
9123 MI.eraseFromParent();
9124 return SinkBB;
9125}
9126
9127static MachineBasicBlock *
9129 const LoongArchSubtarget &Subtarget) {
9130 unsigned InsOp;
9131 unsigned BroadcastOp;
9132 unsigned HalfSize;
9133 switch (MI.getOpcode()) {
9134 default:
9135 llvm_unreachable("Unexpected opcode");
9136 case LoongArch::PseudoXVINSGR2VR_B:
9137 HalfSize = 16;
9138 BroadcastOp = LoongArch::XVREPLGR2VR_B;
9139 InsOp = LoongArch::XVEXTRINS_B;
9140 break;
9141 case LoongArch::PseudoXVINSGR2VR_H:
9142 HalfSize = 8;
9143 BroadcastOp = LoongArch::XVREPLGR2VR_H;
9144 InsOp = LoongArch::XVEXTRINS_H;
9145 break;
9146 }
9147 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
9148 const TargetRegisterClass *RC = &LoongArch::LASX256RegClass;
9149 const TargetRegisterClass *SubRC = &LoongArch::LSX128RegClass;
9150 DebugLoc DL = MI.getDebugLoc();
9152 // XDst = vector_insert XSrc, Elt, Idx
9153 Register XDst = MI.getOperand(0).getReg();
9154 Register XSrc = MI.getOperand(1).getReg();
9155 Register Elt = MI.getOperand(2).getReg();
9156 unsigned Idx = MI.getOperand(3).getImm();
9157
9158 if (XSrc.isVirtual() && MRI.getVRegDef(XSrc)->isImplicitDef() &&
9159 Idx < HalfSize) {
9160 Register ScratchSubReg1 = MRI.createVirtualRegister(SubRC);
9161 Register ScratchSubReg2 = MRI.createVirtualRegister(SubRC);
9162
9163 BuildMI(*BB, MI, DL, TII->get(LoongArch::COPY), ScratchSubReg1)
9164 .addReg(XSrc, {}, LoongArch::sub_128);
9165 BuildMI(*BB, MI, DL,
9166 TII->get(HalfSize == 8 ? LoongArch::VINSGR2VR_H
9167 : LoongArch::VINSGR2VR_B),
9168 ScratchSubReg2)
9169 .addReg(ScratchSubReg1)
9170 .addReg(Elt)
9171 .addImm(Idx);
9172
9173 BuildMI(*BB, MI, DL, TII->get(LoongArch::SUBREG_TO_REG), XDst)
9174 .addReg(ScratchSubReg2)
9175 .addImm(LoongArch::sub_128);
9176 } else {
9177 Register ScratchReg1 = MRI.createVirtualRegister(RC);
9178 Register ScratchReg2 = MRI.createVirtualRegister(RC);
9179
9180 BuildMI(*BB, MI, DL, TII->get(BroadcastOp), ScratchReg1).addReg(Elt);
9181
9182 BuildMI(*BB, MI, DL, TII->get(LoongArch::XVPERMI_Q), ScratchReg2)
9183 .addReg(ScratchReg1)
9184 .addReg(XSrc)
9185 .addImm(Idx >= HalfSize ? 48 : 18);
9186
9187 BuildMI(*BB, MI, DL, TII->get(InsOp), XDst)
9188 .addReg(XSrc)
9189 .addReg(ScratchReg2)
9190 .addImm((Idx >= HalfSize ? Idx - HalfSize : Idx) * 17);
9191 }
9192
9193 MI.eraseFromParent();
9194 return BB;
9195}
9196
9199 const LoongArchSubtarget &Subtarget) {
9200 assert(Subtarget.hasExtLSX());
9201 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
9202 const TargetRegisterClass *RC = &LoongArch::LSX128RegClass;
9203 DebugLoc DL = MI.getDebugLoc();
9205 Register Dst = MI.getOperand(0).getReg();
9206 Register Src = MI.getOperand(1).getReg();
9207
9208 unsigned BroadcastOp, CTOp, PickOp;
9209 switch (MI.getOpcode()) {
9210 default:
9211 llvm_unreachable("Unexpected opcode");
9212 case LoongArch::PseudoCTPOP_B:
9213 BroadcastOp = LoongArch::VREPLGR2VR_B;
9214 CTOp = LoongArch::VPCNT_B;
9215 PickOp = LoongArch::VPICKVE2GR_B;
9216 break;
9217 case LoongArch::PseudoCTPOP_H:
9218 case LoongArch::PseudoCTPOP_H_LA32:
9219 BroadcastOp = LoongArch::VREPLGR2VR_H;
9220 CTOp = LoongArch::VPCNT_H;
9221 PickOp = LoongArch::VPICKVE2GR_H;
9222 break;
9223 case LoongArch::PseudoCTPOP_W:
9224 case LoongArch::PseudoCTPOP_W_LA32:
9225 BroadcastOp = LoongArch::VREPLGR2VR_W;
9226 CTOp = LoongArch::VPCNT_W;
9227 PickOp = LoongArch::VPICKVE2GR_W;
9228 break;
9229 case LoongArch::PseudoCTPOP_D:
9230 BroadcastOp = LoongArch::VREPLGR2VR_D;
9231 CTOp = LoongArch::VPCNT_D;
9232 PickOp = LoongArch::VPICKVE2GR_D;
9233 break;
9234 }
9235
9236 Register ScratchReg1 = MRI.createVirtualRegister(RC);
9237 Register ScratchReg2 = MRI.createVirtualRegister(RC);
9238 BuildMI(*BB, MI, DL, TII->get(BroadcastOp), ScratchReg1).addReg(Src);
9239 BuildMI(*BB, MI, DL, TII->get(CTOp), ScratchReg2).addReg(ScratchReg1);
9240 BuildMI(*BB, MI, DL, TII->get(PickOp), Dst).addReg(ScratchReg2).addImm(0);
9241
9242 MI.eraseFromParent();
9243 return BB;
9244}
9245
9246static MachineBasicBlock *
9248 const LoongArchSubtarget &Subtarget) {
9249 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
9250 const TargetRegisterClass *RC = &LoongArch::LSX128RegClass;
9251 const LoongArchRegisterInfo *TRI = Subtarget.getRegisterInfo();
9253 Register Dst = MI.getOperand(0).getReg();
9254 Register Src = MI.getOperand(1).getReg();
9255 DebugLoc DL = MI.getDebugLoc();
9256 unsigned EleBits = 8;
9257 unsigned NotOpc = 0;
9258 unsigned MskOpc;
9259
9260 switch (MI.getOpcode()) {
9261 default:
9262 llvm_unreachable("Unexpected opcode");
9263 case LoongArch::PseudoVMSKLTZ_B:
9264 MskOpc = LoongArch::VMSKLTZ_B;
9265 break;
9266 case LoongArch::PseudoVMSKLTZ_H:
9267 MskOpc = LoongArch::VMSKLTZ_H;
9268 EleBits = 16;
9269 break;
9270 case LoongArch::PseudoVMSKLTZ_W:
9271 MskOpc = LoongArch::VMSKLTZ_W;
9272 EleBits = 32;
9273 break;
9274 case LoongArch::PseudoVMSKLTZ_D:
9275 MskOpc = LoongArch::VMSKLTZ_D;
9276 EleBits = 64;
9277 break;
9278 case LoongArch::PseudoVMSKGEZ_B:
9279 MskOpc = LoongArch::VMSKGEZ_B;
9280 break;
9281 case LoongArch::PseudoVMSKEQZ_B:
9282 MskOpc = LoongArch::VMSKNZ_B;
9283 NotOpc = LoongArch::VNOR_V;
9284 break;
9285 case LoongArch::PseudoVMSKNEZ_B:
9286 MskOpc = LoongArch::VMSKNZ_B;
9287 break;
9288 case LoongArch::PseudoXVMSKLTZ_B:
9289 MskOpc = LoongArch::XVMSKLTZ_B;
9290 RC = &LoongArch::LASX256RegClass;
9291 break;
9292 case LoongArch::PseudoXVMSKLTZ_H:
9293 MskOpc = LoongArch::XVMSKLTZ_H;
9294 RC = &LoongArch::LASX256RegClass;
9295 EleBits = 16;
9296 break;
9297 case LoongArch::PseudoXVMSKLTZ_W:
9298 MskOpc = LoongArch::XVMSKLTZ_W;
9299 RC = &LoongArch::LASX256RegClass;
9300 EleBits = 32;
9301 break;
9302 case LoongArch::PseudoXVMSKLTZ_D:
9303 MskOpc = LoongArch::XVMSKLTZ_D;
9304 RC = &LoongArch::LASX256RegClass;
9305 EleBits = 64;
9306 break;
9307 case LoongArch::PseudoXVMSKGEZ_B:
9308 MskOpc = LoongArch::XVMSKGEZ_B;
9309 RC = &LoongArch::LASX256RegClass;
9310 break;
9311 case LoongArch::PseudoXVMSKEQZ_B:
9312 MskOpc = LoongArch::XVMSKNZ_B;
9313 NotOpc = LoongArch::XVNOR_V;
9314 RC = &LoongArch::LASX256RegClass;
9315 break;
9316 case LoongArch::PseudoXVMSKNEZ_B:
9317 MskOpc = LoongArch::XVMSKNZ_B;
9318 RC = &LoongArch::LASX256RegClass;
9319 break;
9320 }
9321
9322 Register Msk = MRI.createVirtualRegister(RC);
9323 if (NotOpc) {
9324 Register Tmp = MRI.createVirtualRegister(RC);
9325 BuildMI(*BB, MI, DL, TII->get(MskOpc), Tmp).addReg(Src);
9326 BuildMI(*BB, MI, DL, TII->get(NotOpc), Msk)
9327 .addReg(Tmp)
9328 .addReg(Tmp);
9329 } else {
9330 BuildMI(*BB, MI, DL, TII->get(MskOpc), Msk).addReg(Src);
9331 }
9332
9333 if (TRI->getRegSizeInBits(*RC) > 128) {
9334 Register Lo = MRI.createVirtualRegister(&LoongArch::GPRRegClass);
9335 Register Hi = MRI.createVirtualRegister(&LoongArch::GPRRegClass);
9336 BuildMI(*BB, MI, DL, TII->get(LoongArch::XVPICKVE2GR_WU), Lo)
9337 .addReg(Msk)
9338 .addImm(0);
9339 BuildMI(*BB, MI, DL, TII->get(LoongArch::XVPICKVE2GR_WU), Hi)
9340 .addReg(Msk)
9341 .addImm(4);
9342 BuildMI(*BB, MI, DL,
9343 TII->get(Subtarget.is64Bit() ? LoongArch::BSTRINS_D
9344 : LoongArch::BSTRINS_W),
9345 Dst)
9346 .addReg(Lo)
9347 .addReg(Hi)
9348 .addImm(256 / EleBits - 1)
9349 .addImm(128 / EleBits);
9350 } else {
9351 BuildMI(*BB, MI, DL, TII->get(LoongArch::VPICKVE2GR_HU), Dst)
9352 .addReg(Msk)
9353 .addImm(0);
9354 }
9355
9356 MI.eraseFromParent();
9357 return BB;
9358}
9359
9360static MachineBasicBlock *
9362 const LoongArchSubtarget &Subtarget) {
9363 assert(MI.getOpcode() == LoongArch::SplitPairF64Pseudo &&
9364 "Unexpected instruction");
9365
9366 MachineFunction &MF = *BB->getParent();
9367 DebugLoc DL = MI.getDebugLoc();
9369 Register LoReg = MI.getOperand(0).getReg();
9370 Register HiReg = MI.getOperand(1).getReg();
9371 Register SrcReg = MI.getOperand(2).getReg();
9372
9373 BuildMI(*BB, MI, DL, TII.get(LoongArch::MOVFR2GR_S_64), LoReg).addReg(SrcReg);
9374 BuildMI(*BB, MI, DL, TII.get(LoongArch::MOVFRH2GR_S), HiReg)
9375 .addReg(SrcReg, getKillRegState(MI.getOperand(2).isKill()));
9376 MI.eraseFromParent(); // The pseudo instruction is gone now.
9377 return BB;
9378}
9379
9380static MachineBasicBlock *
9382 const LoongArchSubtarget &Subtarget) {
9383 assert(MI.getOpcode() == LoongArch::BuildPairF64Pseudo &&
9384 "Unexpected instruction");
9385
9386 MachineFunction &MF = *BB->getParent();
9387 DebugLoc DL = MI.getDebugLoc();
9390 Register TmpReg = MRI.createVirtualRegister(&LoongArch::FPR64RegClass);
9391 Register DstReg = MI.getOperand(0).getReg();
9392 Register LoReg = MI.getOperand(1).getReg();
9393 Register HiReg = MI.getOperand(2).getReg();
9394
9395 BuildMI(*BB, MI, DL, TII.get(LoongArch::MOVGR2FR_W_64), TmpReg)
9396 .addReg(LoReg, getKillRegState(MI.getOperand(1).isKill()));
9397 BuildMI(*BB, MI, DL, TII.get(LoongArch::MOVGR2FRH_W), DstReg)
9398 .addReg(TmpReg)
9399 .addReg(HiReg, getKillRegState(MI.getOperand(2).isKill()));
9400 MI.eraseFromParent(); // The pseudo instruction is gone now.
9401 return BB;
9402}
9403
9405 switch (MI.getOpcode()) {
9406 default:
9407 return false;
9408 case LoongArch::Select_GPR_Using_CC_GPR:
9409 return true;
9410 }
9411}
9412
9413static MachineBasicBlock *
9415 const LoongArchSubtarget &Subtarget) {
9416 // To "insert" Select_* instructions, we actually have to insert the triangle
9417 // control-flow pattern. The incoming instructions know the destination vreg
9418 // to set, the condition code register to branch on, the true/false values to
9419 // select between, and the condcode to use to select the appropriate branch.
9420 //
9421 // We produce the following control flow:
9422 // HeadMBB
9423 // | \
9424 // | IfFalseMBB
9425 // | /
9426 // TailMBB
9427 //
9428 // When we find a sequence of selects we attempt to optimize their emission
9429 // by sharing the control flow. Currently we only handle cases where we have
9430 // multiple selects with the exact same condition (same LHS, RHS and CC).
9431 // The selects may be interleaved with other instructions if the other
9432 // instructions meet some requirements we deem safe:
9433 // - They are not pseudo instructions.
9434 // - They are debug instructions. Otherwise,
9435 // - They do not have side-effects, do not access memory and their inputs do
9436 // not depend on the results of the select pseudo-instructions.
9437 // The TrueV/FalseV operands of the selects cannot depend on the result of
9438 // previous selects in the sequence.
9439 // These conditions could be further relaxed. See the X86 target for a
9440 // related approach and more information.
9441
9442 Register LHS = MI.getOperand(1).getReg();
9443 Register RHS;
9444 if (MI.getOperand(2).isReg())
9445 RHS = MI.getOperand(2).getReg();
9446 auto CC = static_cast<unsigned>(MI.getOperand(3).getImm());
9447
9448 SmallVector<MachineInstr *, 4> SelectDebugValues;
9449 SmallSet<Register, 4> SelectDests;
9450 SelectDests.insert(MI.getOperand(0).getReg());
9451
9452 MachineInstr *LastSelectPseudo = &MI;
9453 for (auto E = BB->end(), SequenceMBBI = MachineBasicBlock::iterator(MI);
9454 SequenceMBBI != E; ++SequenceMBBI) {
9455 if (SequenceMBBI->isDebugInstr())
9456 continue;
9457 if (isSelectPseudo(*SequenceMBBI)) {
9458 if (SequenceMBBI->getOperand(1).getReg() != LHS ||
9459 !SequenceMBBI->getOperand(2).isReg() ||
9460 SequenceMBBI->getOperand(2).getReg() != RHS ||
9461 SequenceMBBI->getOperand(3).getImm() != CC ||
9462 SelectDests.count(SequenceMBBI->getOperand(4).getReg()) ||
9463 SelectDests.count(SequenceMBBI->getOperand(5).getReg()))
9464 break;
9465 LastSelectPseudo = &*SequenceMBBI;
9466 SequenceMBBI->collectDebugValues(SelectDebugValues);
9467 SelectDests.insert(SequenceMBBI->getOperand(0).getReg());
9468 continue;
9469 }
9470 if (SequenceMBBI->hasUnmodeledSideEffects() ||
9471 SequenceMBBI->mayLoadOrStore() ||
9472 SequenceMBBI->usesCustomInsertionHook())
9473 break;
9474 if (llvm::any_of(SequenceMBBI->operands(), [&](MachineOperand &MO) {
9475 return MO.isReg() && MO.isUse() && SelectDests.count(MO.getReg());
9476 }))
9477 break;
9478 }
9479
9480 const LoongArchInstrInfo &TII = *Subtarget.getInstrInfo();
9481 const BasicBlock *LLVM_BB = BB->getBasicBlock();
9482 DebugLoc DL = MI.getDebugLoc();
9484
9485 MachineBasicBlock *HeadMBB = BB;
9486 MachineFunction *F = BB->getParent();
9487 MachineBasicBlock *TailMBB = F->CreateMachineBasicBlock(LLVM_BB);
9488 MachineBasicBlock *IfFalseMBB = F->CreateMachineBasicBlock(LLVM_BB);
9489
9490 F->insert(I, IfFalseMBB);
9491 F->insert(I, TailMBB);
9492
9493 // Set the call frame size on entry to the new basic blocks.
9494 unsigned CallFrameSize = TII.getCallFrameSizeAt(*LastSelectPseudo);
9495 IfFalseMBB->setCallFrameSize(CallFrameSize);
9496 TailMBB->setCallFrameSize(CallFrameSize);
9497
9498 // Transfer debug instructions associated with the selects to TailMBB.
9499 for (MachineInstr *DebugInstr : SelectDebugValues) {
9500 TailMBB->push_back(DebugInstr->removeFromParent());
9501 }
9502
9503 // Move all instructions after the sequence to TailMBB.
9504 TailMBB->splice(TailMBB->end(), HeadMBB,
9505 std::next(LastSelectPseudo->getIterator()), HeadMBB->end());
9506 // Update machine-CFG edges by transferring all successors of the current
9507 // block to the new block which will contain the Phi nodes for the selects.
9508 TailMBB->transferSuccessorsAndUpdatePHIs(HeadMBB);
9509 // Set the successors for HeadMBB.
9510 HeadMBB->addSuccessor(IfFalseMBB);
9511 HeadMBB->addSuccessor(TailMBB);
9512
9513 // Insert appropriate branch.
9514 if (MI.getOperand(2).isImm())
9515 BuildMI(HeadMBB, DL, TII.get(CC))
9516 .addReg(LHS)
9517 .addImm(MI.getOperand(2).getImm())
9518 .addMBB(TailMBB);
9519 else
9520 BuildMI(HeadMBB, DL, TII.get(CC)).addReg(LHS).addReg(RHS).addMBB(TailMBB);
9521
9522 // IfFalseMBB just falls through to TailMBB.
9523 IfFalseMBB->addSuccessor(TailMBB);
9524
9525 // Create PHIs for all of the select pseudo-instructions.
9526 auto SelectMBBI = MI.getIterator();
9527 auto SelectEnd = std::next(LastSelectPseudo->getIterator());
9528 auto InsertionPoint = TailMBB->begin();
9529 while (SelectMBBI != SelectEnd) {
9530 auto Next = std::next(SelectMBBI);
9531 if (isSelectPseudo(*SelectMBBI)) {
9532 // %Result = phi [ %TrueValue, HeadMBB ], [ %FalseValue, IfFalseMBB ]
9533 BuildMI(*TailMBB, InsertionPoint, SelectMBBI->getDebugLoc(),
9534 TII.get(LoongArch::PHI), SelectMBBI->getOperand(0).getReg())
9535 .addReg(SelectMBBI->getOperand(4).getReg())
9536 .addMBB(HeadMBB)
9537 .addReg(SelectMBBI->getOperand(5).getReg())
9538 .addMBB(IfFalseMBB);
9539 SelectMBBI->eraseFromParent();
9540 }
9541 SelectMBBI = Next;
9542 }
9543
9544 F->getProperties().resetNoPHIs();
9545 return TailMBB;
9546}
9547
9548MachineBasicBlock *LoongArchTargetLowering::EmitInstrWithCustomInserter(
9549 MachineInstr &MI, MachineBasicBlock *BB) const {
9550 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
9551 DebugLoc DL = MI.getDebugLoc();
9552
9553 switch (MI.getOpcode()) {
9554 default:
9555 llvm_unreachable("Unexpected instr type to insert");
9556 case LoongArch::DIV_W:
9557 case LoongArch::DIV_WU:
9558 case LoongArch::MOD_W:
9559 case LoongArch::MOD_WU:
9560 case LoongArch::DIV_D:
9561 case LoongArch::DIV_DU:
9562 case LoongArch::MOD_D:
9563 case LoongArch::MOD_DU:
9564 return insertDivByZeroTrap(MI, BB);
9565 break;
9566 case LoongArch::WRFCSR: {
9567 BuildMI(*BB, MI, DL, TII->get(LoongArch::MOVGR2FCSR),
9568 LoongArch::FCSR0 + MI.getOperand(0).getImm())
9569 .addReg(MI.getOperand(1).getReg());
9570 MI.eraseFromParent();
9571 return BB;
9572 }
9573 case LoongArch::RDFCSR: {
9574 MachineInstr *ReadFCSR =
9575 BuildMI(*BB, MI, DL, TII->get(LoongArch::MOVFCSR2GR),
9576 MI.getOperand(0).getReg())
9577 .addReg(LoongArch::FCSR0 + MI.getOperand(1).getImm());
9578 ReadFCSR->getOperand(1).setIsUndef();
9579 MI.eraseFromParent();
9580 return BB;
9581 }
9582 case LoongArch::Select_GPR_Using_CC_GPR:
9583 return emitSelectPseudo(MI, BB, Subtarget);
9584 case LoongArch::BuildPairF64Pseudo:
9585 return emitBuildPairF64Pseudo(MI, BB, Subtarget);
9586 case LoongArch::SplitPairF64Pseudo:
9587 return emitSplitPairF64Pseudo(MI, BB, Subtarget);
9588 case LoongArch::PseudoVBZ:
9589 case LoongArch::PseudoVBZ_B:
9590 case LoongArch::PseudoVBZ_H:
9591 case LoongArch::PseudoVBZ_W:
9592 case LoongArch::PseudoVBZ_D:
9593 case LoongArch::PseudoVBNZ:
9594 case LoongArch::PseudoVBNZ_B:
9595 case LoongArch::PseudoVBNZ_H:
9596 case LoongArch::PseudoVBNZ_W:
9597 case LoongArch::PseudoVBNZ_D:
9598 case LoongArch::PseudoXVBZ:
9599 case LoongArch::PseudoXVBZ_B:
9600 case LoongArch::PseudoXVBZ_H:
9601 case LoongArch::PseudoXVBZ_W:
9602 case LoongArch::PseudoXVBZ_D:
9603 case LoongArch::PseudoXVBNZ:
9604 case LoongArch::PseudoXVBNZ_B:
9605 case LoongArch::PseudoXVBNZ_H:
9606 case LoongArch::PseudoXVBNZ_W:
9607 case LoongArch::PseudoXVBNZ_D:
9608 return emitVecCondBranchPseudo(MI, BB, Subtarget);
9609 case LoongArch::PseudoXVINSGR2VR_B:
9610 case LoongArch::PseudoXVINSGR2VR_H:
9611 return emitPseudoXVINSGR2VR(MI, BB, Subtarget);
9612 case LoongArch::PseudoCTPOP_B:
9613 case LoongArch::PseudoCTPOP_H:
9614 case LoongArch::PseudoCTPOP_W:
9615 case LoongArch::PseudoCTPOP_D:
9616 case LoongArch::PseudoCTPOP_H_LA32:
9617 case LoongArch::PseudoCTPOP_W_LA32:
9618 return emitPseudoCTPOP(MI, BB, Subtarget);
9619 case LoongArch::PseudoVMSKLTZ_B:
9620 case LoongArch::PseudoVMSKLTZ_H:
9621 case LoongArch::PseudoVMSKLTZ_W:
9622 case LoongArch::PseudoVMSKLTZ_D:
9623 case LoongArch::PseudoVMSKGEZ_B:
9624 case LoongArch::PseudoVMSKEQZ_B:
9625 case LoongArch::PseudoVMSKNEZ_B:
9626 case LoongArch::PseudoXVMSKLTZ_B:
9627 case LoongArch::PseudoXVMSKLTZ_H:
9628 case LoongArch::PseudoXVMSKLTZ_W:
9629 case LoongArch::PseudoXVMSKLTZ_D:
9630 case LoongArch::PseudoXVMSKGEZ_B:
9631 case LoongArch::PseudoXVMSKEQZ_B:
9632 case LoongArch::PseudoXVMSKNEZ_B:
9633 return emitPseudoVMSKCOND(MI, BB, Subtarget);
9634 case TargetOpcode::STATEPOINT:
9635 // STATEPOINT is a pseudo instruction which has no implicit defs/uses
9636 // while bl call instruction (where statepoint will be lowered at the
9637 // end) has implicit def. This def is early-clobber as it will be set at
9638 // the moment of the call and earlier than any use is read.
9639 // Add this implicit dead def here as a workaround.
9640 MI.addOperand(*MI.getMF(),
9642 LoongArch::R1, /*isDef*/ true,
9643 /*isImp*/ true, /*isKill*/ false, /*isDead*/ true,
9644 /*isUndef*/ false, /*isEarlyClobber*/ true));
9645 if (!Subtarget.is64Bit())
9646 report_fatal_error("STATEPOINT is only supported on 64-bit targets");
9647 return emitPatchPoint(MI, BB);
9648 case LoongArch::PROBED_STACKALLOC_DYN:
9649 return emitDynamicProbedAlloc(MI, BB);
9650 }
9651}
9652
9654 EVT VT, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags,
9655 unsigned *Fast) const {
9656 if (!Subtarget.hasUAL())
9657 return false;
9658
9659 // TODO: set reasonable speed number.
9660 if (Fast)
9661 *Fast = 1;
9662 return true;
9663}
9664
9665//===----------------------------------------------------------------------===//
9666// Calling Convention Implementation
9667//===----------------------------------------------------------------------===//
9668
9669// Eight general-purpose registers a0-a7 used for passing integer arguments,
9670// with a0-a1 reused to return values. Generally, the GPRs are used to pass
9671// fixed-point arguments, and floating-point arguments when no FPR is available
9672// or with soft float ABI.
9673const MCPhysReg ArgGPRs[] = {LoongArch::R4, LoongArch::R5, LoongArch::R6,
9674 LoongArch::R7, LoongArch::R8, LoongArch::R9,
9675 LoongArch::R10, LoongArch::R11};
9676
9677// PreserveNone calling convention:
9678// Arguments may be passed in any general-purpose registers except:
9679// - R1 : return address register
9680// - R22 : frame pointer
9681// - R31 : base pointer
9682//
9683// All general-purpose registers are treated as caller-saved,
9684// except R1 (RA) and R22 (FP).
9685//
9686// Non-volatile registers are allocated first so that a function
9687// can call normal functions without having to spill and reload
9688// argument registers.
9690 LoongArch::R23, LoongArch::R24, LoongArch::R25, LoongArch::R26,
9691 LoongArch::R27, LoongArch::R28, LoongArch::R29, LoongArch::R30,
9692 LoongArch::R4, LoongArch::R5, LoongArch::R6, LoongArch::R7,
9693 LoongArch::R8, LoongArch::R9, LoongArch::R10, LoongArch::R11,
9694 LoongArch::R12, LoongArch::R13, LoongArch::R14, LoongArch::R15,
9695 LoongArch::R16, LoongArch::R17, LoongArch::R18, LoongArch::R19,
9696 LoongArch::R20};
9697
9698// Eight floating-point registers fa0-fa7 used for passing floating-point
9699// arguments, and fa0-fa1 are also used to return values.
9700const MCPhysReg ArgFPR32s[] = {LoongArch::F0, LoongArch::F1, LoongArch::F2,
9701 LoongArch::F3, LoongArch::F4, LoongArch::F5,
9702 LoongArch::F6, LoongArch::F7};
9703// FPR32 and FPR64 alias each other.
9705 LoongArch::F0_64, LoongArch::F1_64, LoongArch::F2_64, LoongArch::F3_64,
9706 LoongArch::F4_64, LoongArch::F5_64, LoongArch::F6_64, LoongArch::F7_64};
9707
9708const MCPhysReg ArgVRs[] = {LoongArch::VR0, LoongArch::VR1, LoongArch::VR2,
9709 LoongArch::VR3, LoongArch::VR4, LoongArch::VR5,
9710 LoongArch::VR6, LoongArch::VR7};
9711
9712const MCPhysReg ArgXRs[] = {LoongArch::XR0, LoongArch::XR1, LoongArch::XR2,
9713 LoongArch::XR3, LoongArch::XR4, LoongArch::XR5,
9714 LoongArch::XR6, LoongArch::XR7};
9715
9717 switch (State.getCallingConv()) {
9719 if (!State.isVarArg())
9720 return State.AllocateReg(PreserveNoneArgGPRs);
9721 [[fallthrough]];
9722 default:
9723 return State.AllocateReg(ArgGPRs);
9724 }
9725}
9726
9727// Pass a 2*GRLen argument that has been split into two GRLen values through
9728// registers or the stack as necessary.
9729static bool CC_LoongArchAssign2GRLen(unsigned GRLen, CCState &State,
9730 CCValAssign VA1, ISD::ArgFlagsTy ArgFlags1,
9731 unsigned ValNo2, MVT ValVT2, MVT LocVT2,
9732 ISD::ArgFlagsTy ArgFlags2) {
9733 unsigned GRLenInBytes = GRLen / 8;
9734 if (Register Reg = allocateArgGPR(State)) {
9735 // At least one half can be passed via register.
9736 State.addLoc(CCValAssign::getReg(VA1.getValNo(), VA1.getValVT(), Reg,
9737 VA1.getLocVT(), CCValAssign::Full));
9738 } else {
9739 // Both halves must be passed on the stack, with proper alignment.
9740 Align StackAlign =
9741 std::max(Align(GRLenInBytes), ArgFlags1.getNonZeroOrigAlign());
9742 State.addLoc(
9744 State.AllocateStack(GRLenInBytes, StackAlign),
9745 VA1.getLocVT(), CCValAssign::Full));
9746 State.addLoc(CCValAssign::getMem(
9747 ValNo2, ValVT2, State.AllocateStack(GRLenInBytes, Align(GRLenInBytes)),
9748 LocVT2, CCValAssign::Full));
9749 return false;
9750 }
9751 if (Register Reg = allocateArgGPR(State)) {
9752 // The second half can also be passed via register.
9753 State.addLoc(
9754 CCValAssign::getReg(ValNo2, ValVT2, Reg, LocVT2, CCValAssign::Full));
9755 } else {
9756 // The second half is passed via the stack, without additional alignment.
9757 State.addLoc(CCValAssign::getMem(
9758 ValNo2, ValVT2, State.AllocateStack(GRLenInBytes, Align(GRLenInBytes)),
9759 LocVT2, CCValAssign::Full));
9760 }
9761 return false;
9762}
9763
9764// Implements the LoongArch calling convention. Returns true upon failure.
9766 unsigned ValNo, MVT ValVT,
9767 CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags,
9768 CCState &State, bool IsRet, Type *OrigTy) {
9769 unsigned GRLen = DL.getLargestLegalIntTypeSizeInBits();
9770 assert((GRLen == 32 || GRLen == 64) && "Unspport GRLen");
9771 MVT GRLenVT = GRLen == 32 ? MVT::i32 : MVT::i64;
9772 MVT LocVT = ValVT;
9773
9774 // Any return value split into more than two values can't be returned
9775 // directly.
9776 if (IsRet && ValNo > 1)
9777 return true;
9778
9779 // If passing a variadic argument, or if no FPR is available.
9780 bool UseGPRForFloat = true;
9781
9782 switch (ABI) {
9783 default:
9784 llvm_unreachable("Unexpected ABI");
9785 break;
9790 UseGPRForFloat = ArgFlags.isVarArg();
9791 break;
9794 break;
9795 }
9796
9797 // If this is a variadic argument, the LoongArch calling convention requires
9798 // that it is assigned an 'even' or 'aligned' register if it has (2*GRLen)/8
9799 // byte alignment. An aligned register should be used regardless of whether
9800 // the original argument was split during legalisation or not. The argument
9801 // will not be passed by registers if the original type is larger than
9802 // 2*GRLen, so the register alignment rule does not apply.
9803 unsigned TwoGRLenInBytes = (2 * GRLen) / 8;
9804 if (ArgFlags.isVarArg() &&
9805 ArgFlags.getNonZeroOrigAlign() == TwoGRLenInBytes &&
9806 DL.getTypeAllocSize(OrigTy) == TwoGRLenInBytes) {
9807 unsigned RegIdx = State.getFirstUnallocated(ArgGPRs);
9808 // Skip 'odd' register if necessary.
9809 if (RegIdx != std::size(ArgGPRs) && RegIdx % 2 == 1)
9810 State.AllocateReg(ArgGPRs);
9811 }
9812
9813 SmallVectorImpl<CCValAssign> &PendingLocs = State.getPendingLocs();
9814 SmallVectorImpl<ISD::ArgFlagsTy> &PendingArgFlags =
9815 State.getPendingArgFlags();
9816
9817 assert(PendingLocs.size() == PendingArgFlags.size() &&
9818 "PendingLocs and PendingArgFlags out of sync");
9819
9820 // FPR32 and FPR64 alias each other.
9821 if (State.getFirstUnallocated(ArgFPR32s) == std::size(ArgFPR32s))
9822 UseGPRForFloat = true;
9823
9824 if (UseGPRForFloat && ValVT == MVT::f32) {
9825 LocVT = GRLenVT;
9826 LocInfo = CCValAssign::BCvt;
9827 } else if (UseGPRForFloat && GRLen == 64 && ValVT == MVT::f64) {
9828 LocVT = MVT::i64;
9829 LocInfo = CCValAssign::BCvt;
9830 } else if (UseGPRForFloat && GRLen == 32 && ValVT == MVT::f64) {
9831 // Handle passing f64 on LA32D with a soft float ABI or when floating point
9832 // registers are exhausted.
9833 assert(PendingLocs.empty() && "Can't lower f64 if it is split");
9834 // Depending on available argument GPRS, f64 may be passed in a pair of
9835 // GPRs, split between a GPR and the stack, or passed completely on the
9836 // stack. LowerCall/LowerFormalArguments/LowerReturn must recognise these
9837 // cases.
9838 MCRegister Reg = allocateArgGPR(State);
9839 if (!Reg) {
9840 int64_t StackOffset = State.AllocateStack(8, Align(8));
9841 State.addLoc(
9842 CCValAssign::getMem(ValNo, ValVT, StackOffset, LocVT, LocInfo));
9843 return false;
9844 }
9845 LocVT = MVT::i32;
9846 State.addLoc(CCValAssign::getCustomReg(ValNo, ValVT, Reg, LocVT, LocInfo));
9847 MCRegister HiReg = allocateArgGPR(State);
9848 if (HiReg) {
9849 State.addLoc(
9850 CCValAssign::getCustomReg(ValNo, ValVT, HiReg, LocVT, LocInfo));
9851 } else {
9852 int64_t StackOffset = State.AllocateStack(4, Align(4));
9853 State.addLoc(
9854 CCValAssign::getCustomMem(ValNo, ValVT, StackOffset, LocVT, LocInfo));
9855 }
9856 return false;
9857 }
9858
9859 // Split arguments might be passed indirectly, so keep track of the pending
9860 // values.
9861 if (ValVT.isScalarInteger() && (ArgFlags.isSplit() || !PendingLocs.empty())) {
9862 LocVT = GRLenVT;
9863 LocInfo = CCValAssign::Indirect;
9864 PendingLocs.push_back(
9865 CCValAssign::getPending(ValNo, ValVT, LocVT, LocInfo));
9866 PendingArgFlags.push_back(ArgFlags);
9867 if (!ArgFlags.isSplitEnd()) {
9868 return false;
9869 }
9870 }
9871
9872 // If the split argument only had two elements, it should be passed directly
9873 // in registers or on the stack.
9874 if (ValVT.isScalarInteger() && ArgFlags.isSplitEnd() &&
9875 PendingLocs.size() <= 2) {
9876 assert(PendingLocs.size() == 2 && "Unexpected PendingLocs.size()");
9877 // Apply the normal calling convention rules to the first half of the
9878 // split argument.
9879 CCValAssign VA = PendingLocs[0];
9880 ISD::ArgFlagsTy AF = PendingArgFlags[0];
9881 PendingLocs.clear();
9882 PendingArgFlags.clear();
9883 return CC_LoongArchAssign2GRLen(GRLen, State, VA, AF, ValNo, ValVT, LocVT,
9884 ArgFlags);
9885 }
9886
9887 // Allocate to a register if possible, or else a stack slot.
9888 Register Reg;
9889 unsigned StoreSizeBytes = GRLen / 8;
9890 Align StackAlign = Align(GRLen / 8);
9891
9892 if (ValVT == MVT::f32 && !UseGPRForFloat) {
9893 Reg = State.AllocateReg(ArgFPR32s);
9894 } else if (ValVT == MVT::f64 && !UseGPRForFloat) {
9895 Reg = State.AllocateReg(ArgFPR64s);
9896 } else if (ValVT.is128BitVector()) {
9897 Reg = State.AllocateReg(ArgVRs);
9898 UseGPRForFloat = false;
9899 StoreSizeBytes = 16;
9900 StackAlign = Align(16);
9901 } else if (ValVT.is256BitVector()) {
9902 Reg = State.AllocateReg(ArgXRs);
9903 UseGPRForFloat = false;
9904 StoreSizeBytes = 32;
9905 StackAlign = Align(32);
9906 } else {
9907 Reg = allocateArgGPR(State);
9908 }
9909
9910 unsigned StackOffset =
9911 Reg ? 0 : State.AllocateStack(StoreSizeBytes, StackAlign);
9912
9913 // If we reach this point and PendingLocs is non-empty, we must be at the
9914 // end of a split argument that must be passed indirectly.
9915 if (!PendingLocs.empty()) {
9916 assert(ArgFlags.isSplitEnd() && "Expected ArgFlags.isSplitEnd()");
9917 assert(PendingLocs.size() > 2 && "Unexpected PendingLocs.size()");
9918 for (auto &It : PendingLocs) {
9919 if (Reg)
9920 It.convertToReg(Reg);
9921 else
9922 It.convertToMem(StackOffset);
9923 State.addLoc(It);
9924 }
9925 PendingLocs.clear();
9926 PendingArgFlags.clear();
9927 return false;
9928 }
9929 assert((!UseGPRForFloat || LocVT == GRLenVT) &&
9930 "Expected an GRLenVT at this stage");
9931
9932 if (Reg) {
9933 State.addLoc(CCValAssign::getReg(ValNo, ValVT, Reg, LocVT, LocInfo));
9934 return false;
9935 }
9936
9937 // When a floating-point value is passed on the stack, no bit-cast is needed.
9938 if (ValVT.isFloatingPoint()) {
9939 LocVT = ValVT;
9940 LocInfo = CCValAssign::Full;
9941 }
9942
9943 State.addLoc(CCValAssign::getMem(ValNo, ValVT, StackOffset, LocVT, LocInfo));
9944 return false;
9945}
9946
9947void LoongArchTargetLowering::analyzeInputArgs(
9948 MachineFunction &MF, CCState &CCInfo,
9949 const SmallVectorImpl<ISD::InputArg> &Ins, bool IsRet,
9950 LoongArchCCAssignFn Fn) const {
9951 FunctionType *FType = MF.getFunction().getFunctionType();
9952 for (unsigned i = 0, e = Ins.size(); i != e; ++i) {
9953 MVT ArgVT = Ins[i].VT;
9954 Type *ArgTy = nullptr;
9955 if (IsRet)
9956 ArgTy = FType->getReturnType();
9957 else if (Ins[i].isOrigArg())
9958 ArgTy = FType->getParamType(Ins[i].getOrigArgIndex());
9960 MF.getSubtarget<LoongArchSubtarget>().getTargetABI();
9961 if (Fn(MF.getDataLayout(), ABI, i, ArgVT, CCValAssign::Full, Ins[i].Flags,
9962 CCInfo, IsRet, ArgTy)) {
9963 LLVM_DEBUG(dbgs() << "InputArg #" << i << " has unhandled type " << ArgVT
9964 << '\n');
9965 llvm_unreachable("");
9966 }
9967 }
9968}
9969
9970void LoongArchTargetLowering::analyzeOutputArgs(
9971 MachineFunction &MF, CCState &CCInfo,
9972 const SmallVectorImpl<ISD::OutputArg> &Outs, bool IsRet,
9973 CallLoweringInfo *CLI, LoongArchCCAssignFn Fn) const {
9974 for (unsigned i = 0, e = Outs.size(); i != e; ++i) {
9975 MVT ArgVT = Outs[i].VT;
9976 Type *OrigTy = CLI ? CLI->getArgs()[Outs[i].OrigArgIndex].Ty : nullptr;
9978 MF.getSubtarget<LoongArchSubtarget>().getTargetABI();
9979 if (Fn(MF.getDataLayout(), ABI, i, ArgVT, CCValAssign::Full, Outs[i].Flags,
9980 CCInfo, IsRet, OrigTy)) {
9981 LLVM_DEBUG(dbgs() << "OutputArg #" << i << " has unhandled type " << ArgVT
9982 << "\n");
9983 llvm_unreachable("");
9984 }
9985 }
9986}
9987
9988// Convert Val to a ValVT. Should not be called for CCValAssign::Indirect
9989// values.
9991 const CCValAssign &VA, const SDLoc &DL) {
9992 switch (VA.getLocInfo()) {
9993 default:
9994 llvm_unreachable("Unexpected CCValAssign::LocInfo");
9995 case CCValAssign::Full:
9997 break;
9998 case CCValAssign::BCvt:
9999 if (VA.getLocVT() == MVT::i64 && VA.getValVT() == MVT::f32)
10000 Val = DAG.getNode(LoongArchISD::MOVGR2FR_W_LA64, DL, MVT::f32, Val);
10001 else
10002 Val = DAG.getNode(ISD::BITCAST, DL, VA.getValVT(), Val);
10003 break;
10004 }
10005 return Val;
10006}
10007
10009 const CCValAssign &VA, const SDLoc &DL,
10010 const ISD::InputArg &In,
10011 const LoongArchTargetLowering &TLI) {
10014 EVT LocVT = VA.getLocVT();
10015 SDValue Val;
10016 const TargetRegisterClass *RC = TLI.getRegClassFor(LocVT.getSimpleVT());
10017 Register VReg = RegInfo.createVirtualRegister(RC);
10018 RegInfo.addLiveIn(VA.getLocReg(), VReg);
10019 Val = DAG.getCopyFromReg(Chain, DL, VReg, LocVT);
10020
10021 // If input is sign extended from 32 bits, note it for the OptW pass.
10022 if (In.isOrigArg()) {
10023 Argument *OrigArg = MF.getFunction().getArg(In.getOrigArgIndex());
10024 if (OrigArg->getType()->isIntegerTy()) {
10025 unsigned BitWidth = OrigArg->getType()->getIntegerBitWidth();
10026 // An input zero extended from i31 can also be considered sign extended.
10027 if ((BitWidth <= 32 && In.Flags.isSExt()) ||
10028 (BitWidth < 32 && In.Flags.isZExt())) {
10031 LAFI->addSExt32Register(VReg);
10032 }
10033 }
10034 }
10035
10036 return convertLocVTToValVT(DAG, Val, VA, DL);
10037}
10038
10039// The caller is responsible for loading the full value if the argument is
10040// passed with CCValAssign::Indirect.
10042 const CCValAssign &VA, const SDLoc &DL) {
10044 MachineFrameInfo &MFI = MF.getFrameInfo();
10045 EVT ValVT = VA.getValVT();
10046 int FI = MFI.CreateFixedObject(ValVT.getStoreSize(), VA.getLocMemOffset(),
10047 /*IsImmutable=*/true);
10048 SDValue FIN = DAG.getFrameIndex(
10050
10051 ISD::LoadExtType ExtType;
10052 switch (VA.getLocInfo()) {
10053 default:
10054 llvm_unreachable("Unexpected CCValAssign::LocInfo");
10055 case CCValAssign::Full:
10057 case CCValAssign::BCvt:
10058 ExtType = ISD::NON_EXTLOAD;
10059 break;
10060 }
10061 return DAG.getExtLoad(
10062 ExtType, DL, VA.getLocVT(), Chain, FIN,
10064}
10065
10067 const CCValAssign &VA,
10068 const CCValAssign &HiVA,
10069 const SDLoc &DL) {
10070 assert(VA.getLocVT() == MVT::i32 && VA.getValVT() == MVT::f64 &&
10071 "Unexpected VA");
10073 MachineFrameInfo &MFI = MF.getFrameInfo();
10075
10076 assert(VA.isRegLoc() && "Expected register VA assignment");
10077
10078 Register LoVReg = RegInfo.createVirtualRegister(&LoongArch::GPRRegClass);
10079 RegInfo.addLiveIn(VA.getLocReg(), LoVReg);
10080 SDValue Lo = DAG.getCopyFromReg(Chain, DL, LoVReg, MVT::i32);
10081 SDValue Hi;
10082 if (HiVA.isMemLoc()) {
10083 // Second half of f64 is passed on the stack.
10084 int FI = MFI.CreateFixedObject(4, HiVA.getLocMemOffset(),
10085 /*IsImmutable=*/true);
10086 SDValue FIN = DAG.getFrameIndex(FI, MVT::i32);
10087 Hi = DAG.getLoad(MVT::i32, DL, Chain, FIN,
10089 } else {
10090 // Second half of f64 is passed in another GPR.
10091 Register HiVReg = RegInfo.createVirtualRegister(&LoongArch::GPRRegClass);
10092 RegInfo.addLiveIn(HiVA.getLocReg(), HiVReg);
10093 Hi = DAG.getCopyFromReg(Chain, DL, HiVReg, MVT::i32);
10094 }
10095 return DAG.getNode(LoongArchISD::BUILD_PAIR_F64, DL, MVT::f64, Lo, Hi);
10096}
10097
10099 const CCValAssign &VA, const SDLoc &DL) {
10100 EVT LocVT = VA.getLocVT();
10101
10102 switch (VA.getLocInfo()) {
10103 default:
10104 llvm_unreachable("Unexpected CCValAssign::LocInfo");
10105 case CCValAssign::Full:
10106 break;
10107 case CCValAssign::BCvt:
10108 if (VA.getLocVT() == MVT::i64 && VA.getValVT() == MVT::f32)
10109 Val = DAG.getNode(LoongArchISD::MOVFR2GR_S_LA64, DL, MVT::i64, Val);
10110 else
10111 Val = DAG.getNode(ISD::BITCAST, DL, LocVT, Val);
10112 break;
10113 }
10114 return Val;
10115}
10116
10117static bool CC_LoongArch_GHC(unsigned ValNo, MVT ValVT, MVT LocVT,
10118 CCValAssign::LocInfo LocInfo,
10119 ISD::ArgFlagsTy ArgFlags, Type *OrigTy,
10120 CCState &State) {
10121 if (LocVT == MVT::i32 || LocVT == MVT::i64) {
10122 // Pass in STG registers: Base, Sp, Hp, R1, R2, R3, R4, R5, SpLim
10123 // s0 s1 s2 s3 s4 s5 s6 s7 s8
10124 static const MCPhysReg GPRList[] = {
10125 LoongArch::R23, LoongArch::R24, LoongArch::R25,
10126 LoongArch::R26, LoongArch::R27, LoongArch::R28,
10127 LoongArch::R29, LoongArch::R30, LoongArch::R31};
10128 if (MCRegister Reg = State.AllocateReg(GPRList)) {
10129 State.addLoc(CCValAssign::getReg(ValNo, ValVT, Reg, LocVT, LocInfo));
10130 return false;
10131 }
10132 }
10133
10134 if (LocVT == MVT::f32) {
10135 // Pass in STG registers: F1, F2, F3, F4
10136 // fs0,fs1,fs2,fs3
10137 static const MCPhysReg FPR32List[] = {LoongArch::F24, LoongArch::F25,
10138 LoongArch::F26, LoongArch::F27};
10139 if (MCRegister Reg = State.AllocateReg(FPR32List)) {
10140 State.addLoc(CCValAssign::getReg(ValNo, ValVT, Reg, LocVT, LocInfo));
10141 return false;
10142 }
10143 }
10144
10145 if (LocVT == MVT::f64) {
10146 // Pass in STG registers: D1, D2, D3, D4
10147 // fs4,fs5,fs6,fs7
10148 static const MCPhysReg FPR64List[] = {LoongArch::F28_64, LoongArch::F29_64,
10149 LoongArch::F30_64, LoongArch::F31_64};
10150 if (MCRegister Reg = State.AllocateReg(FPR64List)) {
10151 State.addLoc(CCValAssign::getReg(ValNo, ValVT, Reg, LocVT, LocInfo));
10152 return false;
10153 }
10154 }
10155
10156 report_fatal_error("No registers left in GHC calling convention");
10157 return true;
10158}
10159
10160// Transform physical registers into virtual registers.
10162 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
10163 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
10164 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
10165
10167
10168 switch (CallConv) {
10169 default:
10170 llvm_unreachable("Unsupported calling convention");
10171 case CallingConv::C:
10172 case CallingConv::Fast:
10175 break;
10176 case CallingConv::GHC:
10177 if (!MF.getSubtarget().hasFeature(LoongArch::FeatureBasicF) ||
10178 !MF.getSubtarget().hasFeature(LoongArch::FeatureBasicD))
10180 "GHC calling convention requires the F and D extensions");
10181 }
10182
10183 const Function &Func = MF.getFunction();
10184 EVT PtrVT = getPointerTy(DAG.getDataLayout());
10185 MVT GRLenVT = Subtarget.getGRLenVT();
10186 unsigned GRLenInBytes = Subtarget.getGRLen() / 8;
10187
10188 // Check if this function has any musttail calls. If so, incoming indirect
10189 // arg pointers must be saved in virtual registers so they survive across
10190 // basic blocks (the SelectionDAG is cleared between BBs). Only do this
10191 // when needed to avoid adding register pressure to non-musttail functions.
10192 bool HasMusttail = llvm::any_of(Func, [](const BasicBlock &BB) {
10193 return llvm::any_of(BB, [](const Instruction &I) {
10194 if (const auto *CI = dyn_cast<CallInst>(&I))
10195 return CI->isMustTailCall();
10196 return false;
10197 });
10198 });
10199 // Used with varargs to acumulate store chains.
10200 std::vector<SDValue> OutChains;
10201
10202 // Assign locations to all of the incoming arguments.
10204 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
10205
10206 if (CallConv == CallingConv::GHC)
10208 else
10209 analyzeInputArgs(MF, CCInfo, Ins, /*IsRet=*/false, CC_LoongArch);
10210
10211 for (unsigned i = 0, e = ArgLocs.size(), InsIdx = 0; i != e; ++i, ++InsIdx) {
10212 CCValAssign &VA = ArgLocs[i];
10213 SDValue ArgValue;
10214 // Passing f64 on LA32D with a soft float ABI must be handled as a special
10215 // case.
10216 if (VA.getLocVT() == MVT::i32 && VA.getValVT() == MVT::f64) {
10217 assert(VA.needsCustom());
10218 ArgValue = unpackF64OnLA32DSoftABI(DAG, Chain, VA, ArgLocs[++i], DL);
10219 } else if (VA.isRegLoc())
10220 ArgValue = unpackFromRegLoc(DAG, Chain, VA, DL, Ins[InsIdx], *this);
10221 else
10222 ArgValue = unpackFromMemLoc(DAG, Chain, VA, DL);
10223 if (VA.getLocInfo() == CCValAssign::Indirect) {
10224 // If the original argument was split and passed by reference, we need to
10225 // load all parts of it here (using the same address).
10226 InVals.push_back(DAG.getLoad(VA.getValVT(), DL, Chain, ArgValue,
10228 unsigned ArgIndex = Ins[InsIdx].OrigArgIndex;
10229 if (HasMusttail) {
10232 Register VReg =
10233 MF.getRegInfo().createVirtualRegister(&LoongArch::GPRRegClass);
10234 Chain = DAG.getCopyToReg(Chain, DL, VReg, ArgValue);
10235 LAFI->setIncomingIndirectArg(ArgIndex, VReg);
10236 }
10237 unsigned ArgPartOffset = Ins[InsIdx].PartOffset;
10238 assert(ArgPartOffset == 0);
10239 while (i + 1 != e && Ins[InsIdx + 1].OrigArgIndex == ArgIndex) {
10240 CCValAssign &PartVA = ArgLocs[i + 1];
10241 unsigned PartOffset = Ins[InsIdx + 1].PartOffset - ArgPartOffset;
10242 SDValue Offset = DAG.getIntPtrConstant(PartOffset, DL);
10243 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, ArgValue, Offset);
10244 InVals.push_back(DAG.getLoad(PartVA.getValVT(), DL, Chain, Address,
10246 ++i;
10247 ++InsIdx;
10248 }
10249 continue;
10250 }
10251 InVals.push_back(ArgValue);
10252 }
10253
10254 if (IsVarArg) {
10256 unsigned Idx = CCInfo.getFirstUnallocated(ArgRegs);
10257 const TargetRegisterClass *RC = &LoongArch::GPRRegClass;
10258 MachineFrameInfo &MFI = MF.getFrameInfo();
10259 MachineRegisterInfo &RegInfo = MF.getRegInfo();
10260 auto *LoongArchFI = MF.getInfo<LoongArchMachineFunctionInfo>();
10261
10262 // Offset of the first variable argument from stack pointer, and size of
10263 // the vararg save area. For now, the varargs save area is either zero or
10264 // large enough to hold a0-a7.
10265 int VaArgOffset, VarArgsSaveSize;
10266
10267 // If all registers are allocated, then all varargs must be passed on the
10268 // stack and we don't need to save any argregs.
10269 if (ArgRegs.size() == Idx) {
10270 VaArgOffset = CCInfo.getStackSize();
10271 VarArgsSaveSize = 0;
10272 } else {
10273 VarArgsSaveSize = GRLenInBytes * (ArgRegs.size() - Idx);
10274 VaArgOffset = -VarArgsSaveSize;
10275 }
10276
10277 // Record the frame index of the first variable argument
10278 // which is a value necessary to VASTART.
10279 int FI = MFI.CreateFixedObject(GRLenInBytes, VaArgOffset, true);
10280 LoongArchFI->setVarArgsFrameIndex(FI);
10281
10282 // If saving an odd number of registers then create an extra stack slot to
10283 // ensure that the frame pointer is 2*GRLen-aligned, which in turn ensures
10284 // offsets to even-numbered registered remain 2*GRLen-aligned.
10285 if (Idx % 2) {
10286 MFI.CreateFixedObject(GRLenInBytes, VaArgOffset - (int)GRLenInBytes,
10287 true);
10288 VarArgsSaveSize += GRLenInBytes;
10289 }
10290
10291 // Copy the integer registers that may have been used for passing varargs
10292 // to the vararg save area.
10293 for (unsigned I = Idx; I < ArgRegs.size();
10294 ++I, VaArgOffset += GRLenInBytes) {
10295 const Register Reg = RegInfo.createVirtualRegister(RC);
10296 RegInfo.addLiveIn(ArgRegs[I], Reg);
10297 SDValue ArgValue = DAG.getCopyFromReg(Chain, DL, Reg, GRLenVT);
10298 FI = MFI.CreateFixedObject(GRLenInBytes, VaArgOffset, true);
10299 SDValue PtrOff = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
10300 SDValue Store = DAG.getStore(Chain, DL, ArgValue, PtrOff,
10302 cast<StoreSDNode>(Store.getNode())
10303 ->getMemOperand()
10304 ->setValue((Value *)nullptr);
10305 OutChains.push_back(Store);
10306 }
10307 LoongArchFI->setVarArgsSaveSize(VarArgsSaveSize);
10308 }
10309
10310 // All stores are grouped in one node to allow the matching between
10311 // the size of Ins and InVals. This only happens for vararg functions.
10312 if (!OutChains.empty()) {
10313 OutChains.push_back(Chain);
10314 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, OutChains);
10315 }
10316
10317 return Chain;
10318}
10319
10321 return CI->isTailCall();
10322}
10323
10324// Check if the return value is used as only a return value, as otherwise
10325// we can't perform a tail-call.
10327 SDValue &Chain) const {
10328 if (N->getNumValues() != 1)
10329 return false;
10330 if (!N->hasNUsesOfValue(1, 0))
10331 return false;
10332
10333 SDNode *Copy = *N->user_begin();
10334 if (Copy->getOpcode() != ISD::CopyToReg)
10335 return false;
10336
10337 // If the ISD::CopyToReg has a glue operand, we conservatively assume it
10338 // isn't safe to perform a tail call.
10339 if (Copy->getGluedNode())
10340 return false;
10341
10342 // The copy must be used by a LoongArchISD::RET, and nothing else.
10343 bool HasRet = false;
10344 for (SDNode *Node : Copy->users()) {
10345 if (Node->getOpcode() != LoongArchISD::RET)
10346 return false;
10347 HasRet = true;
10348 }
10349
10350 if (!HasRet)
10351 return false;
10352
10353 Chain = Copy->getOperand(0);
10354 return true;
10355}
10356
10357// Check whether the call is eligible for tail call optimization.
10358bool LoongArchTargetLowering::isEligibleForTailCallOptimization(
10359 CCState &CCInfo, CallLoweringInfo &CLI, MachineFunction &MF,
10360 const SmallVectorImpl<CCValAssign> &ArgLocs) const {
10361
10362 auto CalleeCC = CLI.CallConv;
10363 auto &Outs = CLI.Outs;
10364 auto &Caller = MF.getFunction();
10365 auto CallerCC = Caller.getCallingConv();
10366
10367 bool IsMustTail = CLI.CB && CLI.CB->isMustTailCall();
10368
10369 // Byval parameters hand the function a pointer directly into the stack area
10370 // we want to reuse during a tail call. Working around this *is* possible
10371 // but less efficient and uglier in LowerCall. For musttail, there is no
10372 // workaround today: a byval arg requires a local copy that becomes invalid
10373 // after the tail call deallocates the caller's frame, so rejecting here
10374 // (and triggering reportFatalInternalError in LowerCall) is safer than
10375 // miscompiling.
10376 for (auto &Arg : Outs)
10377 if (Arg.Flags.isByVal())
10378 return false;
10379
10380 // musttail bypasses the remaining checks: the checks either reject cases
10381 // we handle specially (indirect args are forwarded via incoming pointers,
10382 // stack-passed args reuse the matching incoming layout, sret is forwarded
10383 // like any other pointer arg) or are optimizations not applicable to
10384 // mandatory tail calls.
10385 if (IsMustTail)
10386 return true;
10387
10388 // Do not tail call opt if the stack is used to pass parameters.
10389 if (CCInfo.getStackSize() != 0)
10390 return false;
10391
10392 // Do not tail call opt if any parameters need to be passed indirectly.
10393 for (auto &VA : ArgLocs)
10394 if (VA.getLocInfo() == CCValAssign::Indirect)
10395 return false;
10396
10397 // Do not tail call opt if either caller or callee uses struct return
10398 // semantics.
10399 auto IsCallerStructRet = Caller.hasStructRetAttr();
10400 auto IsCalleeStructRet = Outs.empty() ? false : Outs[0].Flags.isSRet();
10401 if (IsCallerStructRet || IsCalleeStructRet)
10402 return false;
10403
10404 // The callee has to preserve all registers the caller needs to preserve.
10405 const LoongArchRegisterInfo *TRI = Subtarget.getRegisterInfo();
10406 const uint32_t *CallerPreserved = TRI->getCallPreservedMask(MF, CallerCC);
10407 if (CalleeCC != CallerCC) {
10408 const uint32_t *CalleePreserved = TRI->getCallPreservedMask(MF, CalleeCC);
10409 if (!TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved))
10410 return false;
10411 }
10412 return true;
10413}
10414
10416 return DAG.getDataLayout().getPrefTypeAlign(
10417 VT.getTypeForEVT(*DAG.getContext()));
10418}
10419
10420// Lower a call to a callseq_start + CALL + callseq_end chain, and add input
10421// and output parameter nodes.
10422SDValue
10424 SmallVectorImpl<SDValue> &InVals) const {
10425 SelectionDAG &DAG = CLI.DAG;
10426 SDLoc &DL = CLI.DL;
10428 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
10430 SDValue Chain = CLI.Chain;
10431 SDValue Callee = CLI.Callee;
10432 CallingConv::ID CallConv = CLI.CallConv;
10433 bool IsVarArg = CLI.IsVarArg;
10434 EVT PtrVT = getPointerTy(DAG.getDataLayout());
10435 MVT GRLenVT = Subtarget.getGRLenVT();
10436 bool &IsTailCall = CLI.IsTailCall;
10437
10439
10440 // Analyze the operands of the call, assigning locations to each operand.
10442 CCState ArgCCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
10443
10444 if (CallConv == CallingConv::GHC)
10445 ArgCCInfo.AnalyzeCallOperands(Outs, CC_LoongArch_GHC);
10446 else
10447 analyzeOutputArgs(MF, ArgCCInfo, Outs, /*IsRet=*/false, &CLI, CC_LoongArch);
10448
10449 // Check if it's really possible to do a tail call.
10450 if (IsTailCall)
10451 IsTailCall = isEligibleForTailCallOptimization(ArgCCInfo, CLI, MF, ArgLocs);
10452
10453 if (IsTailCall)
10454 ++NumTailCalls;
10455 else if (CLI.CB && CLI.CB->isMustTailCall())
10456 report_fatal_error("failed to perform tail call elimination on a call "
10457 "site marked musttail");
10458
10459 // Get a count of how many bytes are to be pushed on the stack.
10460 unsigned NumBytes = ArgCCInfo.getStackSize();
10461
10462 // Create local copies for byval args.
10463 SmallVector<SDValue> ByValArgs;
10464 for (unsigned i = 0, e = Outs.size(); i != e; ++i) {
10465 ISD::ArgFlagsTy Flags = Outs[i].Flags;
10466 if (!Flags.isByVal())
10467 continue;
10468
10469 SDValue Arg = OutVals[i];
10470 unsigned Size = Flags.getByValSize();
10471 Align Alignment = Flags.getNonZeroByValAlign();
10472
10473 int FI =
10474 MF.getFrameInfo().CreateStackObject(Size, Alignment, /*isSS=*/false);
10475 SDValue FIPtr = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
10476 SDValue SizeNode = DAG.getConstant(Size, DL, GRLenVT);
10477
10478 Chain = DAG.getMemcpy(Chain, DL, FIPtr, Arg, SizeNode, Alignment, Alignment,
10479 /*IsVolatile=*/false,
10480 /*AlwaysInline=*/false, /*CI=*/nullptr, std::nullopt,
10482 ByValArgs.push_back(FIPtr);
10483 }
10484
10485 if (!IsTailCall)
10486 Chain = DAG.getCALLSEQ_START(Chain, NumBytes, 0, CLI.DL);
10487
10488 // Copy argument values to their designated locations.
10490 SmallVector<SDValue> MemOpChains;
10491 SDValue StackPtr;
10492 for (unsigned i = 0, j = 0, e = ArgLocs.size(), OutIdx = 0; i != e;
10493 ++i, ++OutIdx) {
10494 CCValAssign &VA = ArgLocs[i];
10495 SDValue ArgValue = OutVals[OutIdx];
10496 ISD::ArgFlagsTy Flags = Outs[OutIdx].Flags;
10497
10498 // Handle passing f64 on LA32D with a soft float ABI as a special case.
10499 if (VA.getLocVT() == MVT::i32 && VA.getValVT() == MVT::f64) {
10500 assert(VA.isRegLoc() && "Expected register VA assignment");
10501 assert(VA.needsCustom());
10502 SDValue SplitF64 =
10503 DAG.getNode(LoongArchISD::SPLIT_PAIR_F64, DL,
10504 DAG.getVTList(MVT::i32, MVT::i32), ArgValue);
10505 SDValue Lo = SplitF64.getValue(0);
10506 SDValue Hi = SplitF64.getValue(1);
10507
10508 Register RegLo = VA.getLocReg();
10509 RegsToPass.push_back(std::make_pair(RegLo, Lo));
10510
10511 // Get the CCValAssign for the Hi part.
10512 CCValAssign &HiVA = ArgLocs[++i];
10513
10514 if (HiVA.isMemLoc()) {
10515 // Second half of f64 is passed on the stack.
10516 if (!StackPtr.getNode())
10517 StackPtr = DAG.getCopyFromReg(Chain, DL, LoongArch::R3, PtrVT);
10519 DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr,
10520 DAG.getIntPtrConstant(HiVA.getLocMemOffset(), DL));
10521 // Emit the store.
10522 MemOpChains.push_back(DAG.getStore(
10523 Chain, DL, Hi, Address,
10525 } else {
10526 // Second half of f64 is passed in another GPR.
10527 Register RegHigh = HiVA.getLocReg();
10528 RegsToPass.push_back(std::make_pair(RegHigh, Hi));
10529 }
10530 continue;
10531 }
10532
10533 // Promote the value if needed.
10534 // For now, only handle fully promoted and indirect arguments.
10535 if (VA.getLocInfo() == CCValAssign::Indirect) {
10536 // For musttail calls, reuse incoming indirect pointers instead of
10537 // creating new stack temporaries. The incoming pointers point to the
10538 // caller's caller's frame, which remains valid after a tail call.
10539 if (IsTailCall && CLI.CB && CLI.CB->isMustTailCall()) {
10542 unsigned CallArgIdx = Outs[OutIdx].OrigArgIndex;
10543
10544 // Resolve which formal parameter is being passed at this call
10545 // position.
10546 //
10547 // FIXME: Ins[].OrigArgIndex is Argument::getArgNo() (unfiltered),
10548 // but Outs[].OrigArgIndex is an index into a filtered arg list
10549 // (empty types removed, via CallLoweringInfo in the target-
10550 // independent layer). IncomingIndirectArgs is keyed by the
10551 // caller's unfiltered Argument::getArgNo(), so we have to walk
10552 // the caller's formals (same filter) to translate the index.
10553 // This target-independent asymmetry should be normalized so
10554 // backends do not need to re-derive the mapping.
10555 //
10556 // Steps:
10557 // 1. Find the call operand at filtered position CallArgIdx.
10558 // 2. If it is an Argument, use getArgNo() directly (same filter
10559 // for caller formals and call operands).
10560 // 3. Otherwise (computed value), walk the caller's formals and
10561 // skip empty types to map the filtered index to getArgNo().
10562 const Argument *FormalArg = nullptr;
10563 unsigned FilteredIdx = 0;
10564 for (const auto &CallArg : CLI.CB->args()) {
10565 if (CallArg->getType()->isEmptyTy())
10566 continue;
10567 if (FilteredIdx == CallArgIdx) {
10568 FormalArg = dyn_cast<Argument>(CallArg);
10569 break;
10570 }
10571 ++FilteredIdx;
10572 }
10573
10574 // For forwarded args, getArgNo() gives the unfiltered index directly.
10575 // For computed args, walk the caller's formals to resolve it.
10576 unsigned FormalArgIdx = CallArgIdx;
10577 if (FormalArg) {
10578 FormalArgIdx = FormalArg->getArgNo();
10579 } else {
10580 FilteredIdx = 0;
10581 for (const auto &Arg : MF.getFunction().args()) {
10582 if (Arg.getType()->isEmptyTy())
10583 continue;
10584 if (FilteredIdx == CallArgIdx) {
10585 FormalArgIdx = Arg.getArgNo();
10586 break;
10587 }
10588 ++FilteredIdx;
10589 }
10590 }
10591
10592 Register VReg = LAFI->getIncomingIndirectArg(FormalArgIdx);
10593 SDValue CopyOp = DAG.getCopyFromReg(Chain, DL, VReg, PtrVT);
10594 // Thread the CopyFromReg output chain through MemOpChains so the
10595 // TokenFactor below sequences the copy with any stores we emit
10596 // for this argument.
10597 MemOpChains.push_back(CopyOp.getValue(1));
10598 SDValue IncomingPtr = CopyOp;
10599
10600 if (!FormalArg) {
10601 // Computed value: store into the incoming indirect pointer for the
10602 // same-position formal parameter (musttail guarantees matching
10603 // prototypes, so types match). The pointer survives the tail call
10604 // since it points to the caller's caller's frame.
10605 //
10606 // The data-flow edge through IncomingPtr already prevents the
10607 // store from being scheduled before the CopyFromReg. Threading
10608 // CopyOp.getValue(1) (the copy's output chain) into the store
10609 // makes that ordering explicit on the chain edge as well, which
10610 // is the convention for memory ops chaining off their producers.
10611 MemOpChains.push_back(
10612 DAG.getStore(CopyOp.getValue(1), DL, ArgValue, IncomingPtr,
10614 // Store any split parts at their respective offsets.
10615 unsigned ArgPartOffset = Outs[OutIdx].PartOffset;
10616 while (i + 1 != e && Outs[OutIdx + 1].OrigArgIndex == CallArgIdx) {
10617 SDValue PartValue = OutVals[OutIdx + 1];
10618 unsigned PartOffset = Outs[OutIdx + 1].PartOffset - ArgPartOffset;
10619 SDValue Offset = DAG.getIntPtrConstant(PartOffset, DL);
10620 SDValue Addr =
10621 DAG.getNode(ISD::ADD, DL, PtrVT, IncomingPtr, Offset);
10622 MemOpChains.push_back(
10623 DAG.getStore(CopyOp.getValue(1), DL, PartValue, Addr,
10625 ++i;
10626 ++OutIdx;
10627 }
10628 }
10629 ArgValue = IncomingPtr;
10630
10631 // Skip any remaining split parts (for forwarded args, they are
10632 // covered by the forwarded pointer).
10633 while (i + 1 != e && Outs[OutIdx + 1].OrigArgIndex == CallArgIdx) {
10634 ++i;
10635 ++OutIdx;
10636 }
10637 } else {
10638 // Store the argument in a stack slot and pass its address.
10639 Align StackAlign =
10640 std::max(getPrefTypeAlign(Outs[OutIdx].ArgVT, DAG),
10641 getPrefTypeAlign(ArgValue.getValueType(), DAG));
10642 TypeSize StoredSize = ArgValue.getValueType().getStoreSize();
10643 // If the original argument was split and passed by reference, we need
10644 // to store the required parts of it here (and pass just one address).
10645 unsigned ArgIndex = Outs[OutIdx].OrigArgIndex;
10646 unsigned ArgPartOffset = Outs[OutIdx].PartOffset;
10647 assert(ArgPartOffset == 0);
10648 // Calculate the total size to store. We don't have access to what we're
10649 // actually storing other than performing the loop and collecting the
10650 // info.
10652 while (i + 1 != e && Outs[OutIdx + 1].OrigArgIndex == ArgIndex) {
10653 SDValue PartValue = OutVals[OutIdx + 1];
10654 unsigned PartOffset = Outs[OutIdx + 1].PartOffset - ArgPartOffset;
10655 SDValue Offset = DAG.getIntPtrConstant(PartOffset, DL);
10656 EVT PartVT = PartValue.getValueType();
10657 StoredSize += PartVT.getStoreSize();
10658 StackAlign = std::max(StackAlign, getPrefTypeAlign(PartVT, DAG));
10659 Parts.push_back(std::make_pair(PartValue, Offset));
10660 ++i;
10661 ++OutIdx;
10662 }
10663 SDValue SpillSlot = DAG.CreateStackTemporary(StoredSize, StackAlign);
10664 int FI = cast<FrameIndexSDNode>(SpillSlot)->getIndex();
10665 MemOpChains.push_back(
10666 DAG.getStore(Chain, DL, ArgValue, SpillSlot,
10668 for (const auto &Part : Parts) {
10669 SDValue PartValue = Part.first;
10670 SDValue PartOffset = Part.second;
10672 DAG.getNode(ISD::ADD, DL, PtrVT, SpillSlot, PartOffset);
10673 MemOpChains.push_back(
10674 DAG.getStore(Chain, DL, PartValue, Address,
10676 }
10677 ArgValue = SpillSlot;
10678 }
10679 } else {
10680 ArgValue = convertValVTToLocVT(DAG, ArgValue, VA, DL);
10681 }
10682
10683 // Use local copy if it is a byval arg.
10684 if (Flags.isByVal())
10685 ArgValue = ByValArgs[j++];
10686
10687 if (VA.isRegLoc()) {
10688 // Queue up the argument copies and emit them at the end.
10689 RegsToPass.push_back(std::make_pair(VA.getLocReg(), ArgValue));
10690 } else {
10691 assert(VA.isMemLoc() && "Argument not register or memory");
10692 assert((!IsTailCall || (CLI.CB && CLI.CB->isMustTailCall())) &&
10693 "Tail call not allowed if stack is used for passing parameters");
10694
10695 // Work out the address of the stack slot.
10696 if (!StackPtr.getNode())
10697 StackPtr = DAG.getCopyFromReg(Chain, DL, LoongArch::R3, PtrVT);
10699 DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr,
10701
10702 // Emit the store.
10703 MemOpChains.push_back(
10704 DAG.getStore(Chain, DL, ArgValue, Address, MachinePointerInfo()));
10705 }
10706 }
10707
10708 // Join the stores, which are independent of one another.
10709 if (!MemOpChains.empty())
10710 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
10711
10712 SDValue Glue;
10713
10714 // Build a sequence of copy-to-reg nodes, chained and glued together.
10715 for (auto &Reg : RegsToPass) {
10716 Chain = DAG.getCopyToReg(Chain, DL, Reg.first, Reg.second, Glue);
10717 Glue = Chain.getValue(1);
10718 }
10719
10720 // If the callee is a GlobalAddress/ExternalSymbol node, turn it into a
10721 // TargetGlobalAddress/TargetExternalSymbol node so that legalize won't
10722 // split it and then direct call can be matched by PseudoCALL_SMALL.
10724 const GlobalValue *GV = S->getGlobal();
10725 unsigned OpFlags = getTargetMachine().shouldAssumeDSOLocal(GV)
10728 Callee = DAG.getTargetGlobalAddress(S->getGlobal(), DL, PtrVT, 0, OpFlags);
10729 } else if (ExternalSymbolSDNode *S = dyn_cast<ExternalSymbolSDNode>(Callee)) {
10730 unsigned OpFlags = getTargetMachine().shouldAssumeDSOLocal(nullptr)
10733 Callee = DAG.getTargetExternalSymbol(S->getSymbol(), PtrVT, OpFlags);
10734 }
10735
10736 // The first call operand is the chain and the second is the target address.
10738 Ops.push_back(Chain);
10739 Ops.push_back(Callee);
10740
10741 // Add argument registers to the end of the list so that they are
10742 // known live into the call.
10743 for (auto &Reg : RegsToPass)
10744 Ops.push_back(DAG.getRegister(Reg.first, Reg.second.getValueType()));
10745
10746 if (!IsTailCall) {
10747 // Add a register mask operand representing the call-preserved registers.
10748 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
10749 const uint32_t *Mask = TRI->getCallPreservedMask(MF, CallConv);
10750 assert(Mask && "Missing call preserved mask for calling convention");
10751 Ops.push_back(DAG.getRegisterMask(Mask));
10752 }
10753
10754 // Glue the call to the argument copies, if any.
10755 if (Glue.getNode())
10756 Ops.push_back(Glue);
10757
10758 // Emit the call.
10759 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
10760 unsigned Op;
10761 switch (DAG.getTarget().getCodeModel()) {
10762 default:
10763 report_fatal_error("Unsupported code model");
10764 case CodeModel::Small:
10765 Op = IsTailCall ? LoongArchISD::TAIL : LoongArchISD::CALL;
10766 break;
10767 case CodeModel::Medium:
10768 Op = IsTailCall ? LoongArchISD::TAIL_MEDIUM : LoongArchISD::CALL_MEDIUM;
10769 break;
10770 case CodeModel::Large:
10771 assert(Subtarget.is64Bit() && "Large code model requires LA64");
10772 Op = IsTailCall ? LoongArchISD::TAIL_LARGE : LoongArchISD::CALL_LARGE;
10773 break;
10774 }
10775
10776 if (IsTailCall) {
10778 SDValue Ret = DAG.getNode(Op, DL, NodeTys, Ops);
10779 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
10780 return Ret;
10781 }
10782
10783 Chain = DAG.getNode(Op, DL, NodeTys, Ops);
10784 DAG.addNoMergeSiteInfo(Chain.getNode(), CLI.NoMerge);
10785 Glue = Chain.getValue(1);
10786
10787 // Mark the end of the call, which is glued to the call itself.
10788 Chain = DAG.getCALLSEQ_END(Chain, NumBytes, 0, Glue, DL);
10789 Glue = Chain.getValue(1);
10790
10791 // Assign locations to each value returned by this call.
10793 CCState RetCCInfo(CallConv, IsVarArg, MF, RVLocs, *DAG.getContext());
10794 analyzeInputArgs(MF, RetCCInfo, Ins, /*IsRet=*/true, CC_LoongArch);
10795
10796 // Copy all of the result registers out of their specified physreg.
10797 for (unsigned i = 0, e = RVLocs.size(); i != e; ++i) {
10798 auto &VA = RVLocs[i];
10799 // Copy the value out.
10800 SDValue RetValue =
10801 DAG.getCopyFromReg(Chain, DL, VA.getLocReg(), VA.getLocVT(), Glue);
10802 // Glue the RetValue to the end of the call sequence.
10803 Chain = RetValue.getValue(1);
10804 Glue = RetValue.getValue(2);
10805
10806 if (VA.getLocVT() == MVT::i32 && VA.getValVT() == MVT::f64) {
10807 assert(VA.needsCustom());
10808 SDValue RetValue2 = DAG.getCopyFromReg(Chain, DL, RVLocs[++i].getLocReg(),
10809 MVT::i32, Glue);
10810 Chain = RetValue2.getValue(1);
10811 Glue = RetValue2.getValue(2);
10812 RetValue = DAG.getNode(LoongArchISD::BUILD_PAIR_F64, DL, MVT::f64,
10813 RetValue, RetValue2);
10814 } else
10815 RetValue = convertLocVTToValVT(DAG, RetValue, VA, DL);
10816
10817 InVals.push_back(RetValue);
10818 }
10819
10820 return Chain;
10821}
10822
10824 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
10825 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
10826 const Type *RetTy) const {
10828 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
10829
10830 for (unsigned i = 0, e = Outs.size(); i != e; ++i) {
10831 LoongArchABI::ABI ABI =
10832 MF.getSubtarget<LoongArchSubtarget>().getTargetABI();
10833 if (CC_LoongArch(MF.getDataLayout(), ABI, i, Outs[i].VT, CCValAssign::Full,
10834 Outs[i].Flags, CCInfo, /*IsRet=*/true, nullptr))
10835 return false;
10836 }
10837 return true;
10838}
10839
10841 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
10843 const SmallVectorImpl<SDValue> &OutVals, const SDLoc &DL,
10844 SelectionDAG &DAG) const {
10845 // Stores the assignment of the return value to a location.
10847
10848 // Info about the registers and stack slot.
10849 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs,
10850 *DAG.getContext());
10851
10852 analyzeOutputArgs(DAG.getMachineFunction(), CCInfo, Outs, /*IsRet=*/true,
10853 nullptr, CC_LoongArch);
10854 if (CallConv == CallingConv::GHC && !RVLocs.empty())
10855 report_fatal_error("GHC functions return void only");
10856 SDValue Glue;
10857 SmallVector<SDValue, 4> RetOps(1, Chain);
10858
10859 // Copy the result values into the output registers.
10860 for (unsigned i = 0, e = RVLocs.size(), OutIdx = 0; i < e; ++i, ++OutIdx) {
10861 SDValue Val = OutVals[OutIdx];
10862 CCValAssign &VA = RVLocs[i];
10863 assert(VA.isRegLoc() && "Can only return in registers!");
10864
10865 if (VA.getLocVT() == MVT::i32 && VA.getValVT() == MVT::f64) {
10866 // Handle returning f64 on LA32D with a soft float ABI.
10867 assert(VA.isRegLoc() && "Expected return via registers");
10868 assert(VA.needsCustom());
10869 SDValue SplitF64 = DAG.getNode(LoongArchISD::SPLIT_PAIR_F64, DL,
10870 DAG.getVTList(MVT::i32, MVT::i32), Val);
10871 SDValue Lo = SplitF64.getValue(0);
10872 SDValue Hi = SplitF64.getValue(1);
10873 Register RegLo = VA.getLocReg();
10874 Register RegHi = RVLocs[++i].getLocReg();
10875
10876 Chain = DAG.getCopyToReg(Chain, DL, RegLo, Lo, Glue);
10877 Glue = Chain.getValue(1);
10878 RetOps.push_back(DAG.getRegister(RegLo, MVT::i32));
10879 Chain = DAG.getCopyToReg(Chain, DL, RegHi, Hi, Glue);
10880 Glue = Chain.getValue(1);
10881 RetOps.push_back(DAG.getRegister(RegHi, MVT::i32));
10882 } else {
10883 // Handle a 'normal' return.
10884 Val = convertValVTToLocVT(DAG, Val, VA, DL);
10885 Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), Val, Glue);
10886
10887 // Guarantee that all emitted copies are stuck together.
10888 Glue = Chain.getValue(1);
10889 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
10890 }
10891 }
10892
10893 RetOps[0] = Chain; // Update chain.
10894
10895 // Add the glue node if we have it.
10896 if (Glue.getNode())
10897 RetOps.push_back(Glue);
10898
10899 return DAG.getNode(LoongArchISD::RET, DL, MVT::Other, RetOps);
10900}
10901
10902// Check if a constant splat can be generated using [x]vldi, where imm[12] == 1.
10903// Note: The following prefixes are excluded:
10904// imm[11:8] == 4'b0000, 4'b0100, 4'b1000
10905// as they can be represented using [x]vrepli.[whb]
10907 const APInt &SplatValue, const unsigned SplatBitSize) const {
10908 uint64_t RequiredImm = 0;
10909 uint64_t V = SplatValue.getZExtValue();
10910 if (SplatBitSize == 16 && !(V & 0x00FF)) {
10911 // 4'b0101
10912 RequiredImm = (0b10101 << 8) | (V >> 8);
10913 return {true, RequiredImm};
10914 } else if (SplatBitSize == 32) {
10915 // 4'b0001
10916 if (!(V & 0xFFFF00FF)) {
10917 RequiredImm = (0b10001 << 8) | (V >> 8);
10918 return {true, RequiredImm};
10919 }
10920 // 4'b0010
10921 if (!(V & 0xFF00FFFF)) {
10922 RequiredImm = (0b10010 << 8) | (V >> 16);
10923 return {true, RequiredImm};
10924 }
10925 // 4'b0011
10926 if (!(V & 0x00FFFFFF)) {
10927 RequiredImm = (0b10011 << 8) | (V >> 24);
10928 return {true, RequiredImm};
10929 }
10930 // 4'b0110
10931 if ((V & 0xFFFF00FF) == 0xFF) {
10932 RequiredImm = (0b10110 << 8) | (V >> 8);
10933 return {true, RequiredImm};
10934 }
10935 // 4'b0111
10936 if ((V & 0xFF00FFFF) == 0xFFFF) {
10937 RequiredImm = (0b10111 << 8) | (V >> 16);
10938 return {true, RequiredImm};
10939 }
10940 // 4'b1010
10941 if ((V & 0x7E07FFFF) == 0x3E000000 || (V & 0x7E07FFFF) == 0x40000000) {
10942 RequiredImm =
10943 (0b11010 << 8) | (((V >> 24) & 0xC0) ^ 0x40) | ((V >> 19) & 0x3F);
10944 return {true, RequiredImm};
10945 }
10946 } else if (SplatBitSize == 64) {
10947 // 4'b1011
10948 if ((V & 0xFFFFFFFF7E07FFFFULL) == 0x3E000000ULL ||
10949 (V & 0xFFFFFFFF7E07FFFFULL) == 0x40000000ULL) {
10950 RequiredImm =
10951 (0b11011 << 8) | (((V >> 24) & 0xC0) ^ 0x40) | ((V >> 19) & 0x3F);
10952 return {true, RequiredImm};
10953 }
10954 // 4'b1100
10955 if ((V & 0x7FC0FFFFFFFFFFFFULL) == 0x4000000000000000ULL ||
10956 (V & 0x7FC0FFFFFFFFFFFFULL) == 0x3FC0000000000000ULL) {
10957 RequiredImm =
10958 (0b11100 << 8) | (((V >> 56) & 0xC0) ^ 0x40) | ((V >> 48) & 0x3F);
10959 return {true, RequiredImm};
10960 }
10961 // 4'b1001
10962 auto sameBitsPreByte = [](uint64_t x) -> std::pair<bool, uint8_t> {
10963 uint8_t res = 0;
10964 for (int i = 0; i < 8; ++i) {
10965 uint8_t byte = x & 0xFF;
10966 if (byte == 0 || byte == 0xFF)
10967 res |= ((byte & 1) << i);
10968 else
10969 return {false, 0};
10970 x >>= 8;
10971 }
10972 return {true, res};
10973 };
10974 auto [IsSame, Suffix] = sameBitsPreByte(V);
10975 if (IsSame) {
10976 RequiredImm = (0b11001 << 8) | Suffix;
10977 return {true, RequiredImm};
10978 }
10979 }
10980 return {false, RequiredImm};
10981}
10982
10984 EVT VT) const {
10985 if (!Subtarget.hasExtLSX())
10986 return false;
10987
10988 if (VT == MVT::f32) {
10989 uint64_t masked = Imm.bitcastToAPInt().getZExtValue() & 0x7e07ffff;
10990 return (masked == 0x3e000000 || masked == 0x40000000);
10991 }
10992
10993 if (VT == MVT::f64) {
10994 uint64_t masked = Imm.bitcastToAPInt().getZExtValue() & 0x7fc0ffffffffffff;
10995 return (masked == 0x3fc0000000000000 || masked == 0x4000000000000000);
10996 }
10997
10998 return false;
10999}
11000
11001bool LoongArchTargetLowering::isFPImmLegal(const APFloat &Imm, EVT VT,
11002 bool ForCodeSize) const {
11003 // TODO: Maybe need more checks here after vector extension is supported.
11004 if (VT == MVT::f32 && !Subtarget.hasBasicF())
11005 return false;
11006 if (VT == MVT::f64 && !Subtarget.hasBasicD())
11007 return false;
11008 return (Imm.isZero() || Imm.isOne() || isFPImmVLDILegal(Imm, VT));
11009}
11010
11012 return true;
11013}
11014
11016 return true;
11017}
11018
11019bool LoongArchTargetLowering::shouldInsertFencesForAtomic(
11020 const Instruction *I) const {
11021 if (!Subtarget.is64Bit())
11022 return isa<LoadInst>(I) || isa<StoreInst>(I);
11023
11024 if (isa<LoadInst>(I))
11025 return true;
11026
11027 // On LA64, atomic store operations with IntegerBitWidth of 32 and 64 do not
11028 // require fences beacuse we can use amswap_db.[w/d].
11029 Type *Ty = I->getOperand(0)->getType();
11030 if (isa<StoreInst>(I) && Ty->isIntegerTy()) {
11031 unsigned Size = Ty->getIntegerBitWidth();
11032 return (Size == 8 || Size == 16);
11033 }
11034
11035 return false;
11036}
11037
11039 LLVMContext &Context,
11040 EVT VT) const {
11041 if (!VT.isVector())
11042 return getPointerTy(DL);
11044}
11045
11047 unsigned AddressSpace, EVT MemVT, const MachineFunction &MF) const {
11048 // Do not merge to float value size (128 or 256 bits) if no implicit
11049 // float attribute is set.
11050 bool NoFloat = MF.getFunction().hasFnAttribute(Attribute::NoImplicitFloat);
11051 unsigned MaxIntSize = Subtarget.is64Bit() ? 64 : 32;
11052 if (NoFloat)
11053 return MemVT.getSizeInBits() <= MaxIntSize;
11054
11055 // Make sure we don't merge greater than our maximum supported vector width.
11056 if (Subtarget.hasExtLASX())
11057 MaxIntSize = 256;
11058 else if (Subtarget.hasExtLSX())
11059 MaxIntSize = 128;
11060
11061 return MemVT.getSizeInBits() <= MaxIntSize;
11062}
11063
11065 EVT VT = Y.getValueType();
11066
11067 if (VT.isVector())
11068 return Subtarget.hasExtLSX() && VT.isInteger();
11069
11070 return VT.isScalarInteger() && !isa<ConstantSDNode>(Y);
11071}
11072
11075 MachineFunction &MF, unsigned Intrinsic) const {
11076 switch (Intrinsic) {
11077 default:
11078 return;
11079 case Intrinsic::loongarch_masked_atomicrmw_xchg_i32:
11080 case Intrinsic::loongarch_masked_atomicrmw_add_i32:
11081 case Intrinsic::loongarch_masked_atomicrmw_sub_i32:
11082 case Intrinsic::loongarch_masked_atomicrmw_nand_i32: {
11083 IntrinsicInfo Info;
11085 Info.memVT = MVT::i32;
11086 Info.ptrVal = I.getArgOperand(0);
11087 Info.offset = 0;
11088 Info.align = Align(4);
11091 Infos.push_back(Info);
11092 return;
11093 // TODO: Add more Intrinsics later.
11094 }
11095 }
11096}
11097
11098// When -mlamcas is enabled, MinCmpXchgSizeInBits will be set to 8,
11099// atomicrmw and/or/xor operations with operands less than 32 bits cannot be
11100// expanded to am{and/or/xor}[_db].w through AtomicExpandPass. To prevent
11101// regression, we need to implement it manually.
11104
11106 Op == AtomicRMWInst::And) &&
11107 "Unable to expand");
11108 unsigned MinWordSize = 4;
11109
11110 IRBuilder<> Builder(AI);
11111 LLVMContext &Ctx = Builder.getContext();
11112 const DataLayout &DL = AI->getDataLayout();
11113 Type *ValueType = AI->getType();
11114 Type *WordType = Type::getIntNTy(Ctx, MinWordSize * 8);
11115
11116 Value *Addr = AI->getPointerOperand();
11117 PointerType *PtrTy = cast<PointerType>(Addr->getType());
11118 IntegerType *IntTy = DL.getIndexType(Ctx, PtrTy->getAddressSpace());
11119
11120 Value *AlignedAddr = Builder.CreateIntrinsic(
11121 Intrinsic::ptrmask, {PtrTy, IntTy},
11122 {Addr, ConstantInt::get(IntTy, ~(uint64_t)(MinWordSize - 1))}, nullptr,
11123 "AlignedAddr");
11124
11125 Value *AddrInt = Builder.CreatePtrToInt(Addr, IntTy);
11126 Value *PtrLSB = Builder.CreateAnd(AddrInt, MinWordSize - 1, "PtrLSB");
11127 Value *ShiftAmt = Builder.CreateShl(PtrLSB, 3);
11128 ShiftAmt = Builder.CreateTrunc(ShiftAmt, WordType, "ShiftAmt");
11129 Value *Mask = Builder.CreateShl(
11130 ConstantInt::get(WordType,
11131 (1 << (DL.getTypeStoreSize(ValueType) * 8)) - 1),
11132 ShiftAmt, "Mask");
11133 Value *Inv_Mask = Builder.CreateNot(Mask, "Inv_Mask");
11134 Value *ValOperand_Shifted =
11135 Builder.CreateShl(Builder.CreateZExt(AI->getValOperand(), WordType),
11136 ShiftAmt, "ValOperand_Shifted");
11137 Value *NewOperand;
11138 if (Op == AtomicRMWInst::And)
11139 NewOperand = Builder.CreateOr(ValOperand_Shifted, Inv_Mask, "AndOperand");
11140 else
11141 NewOperand = ValOperand_Shifted;
11142
11143 AtomicRMWInst *NewAI =
11144 Builder.CreateAtomicRMW(Op, AlignedAddr, NewOperand, Align(MinWordSize),
11145 AI->getOrdering(), AI->getSyncScopeID());
11146
11147 Value *Shift = Builder.CreateLShr(NewAI, ShiftAmt, "shifted");
11148 Value *Trunc = Builder.CreateTrunc(Shift, ValueType, "extracted");
11149 Value *FinalOldResult = Builder.CreateBitCast(Trunc, ValueType);
11150 AI->replaceAllUsesWith(FinalOldResult);
11151 AI->eraseFromParent();
11152}
11153
11156 const AtomicRMWInst *AI) const {
11157 // TODO: Add more AtomicRMWInst that needs to be extended.
11158
11159 // Since floating-point operation requires a non-trivial set of data
11160 // operations, use CmpXChg to expand.
11161 if (AI->isFloatingPointOperation() ||
11167
11168 if (Subtarget.hasLAM_BH() && Subtarget.is64Bit() &&
11171 AI->getOperation() == AtomicRMWInst::Sub)) {
11173 }
11174
11175 unsigned Size = AI->getType()->getPrimitiveSizeInBits();
11176 if (Subtarget.hasLAMCAS()) {
11177 if (Size < 32 && (AI->getOperation() == AtomicRMWInst::And ||
11181 if (AI->getOperation() == AtomicRMWInst::Nand || Size < 32)
11183 }
11184
11185 if (Size == 8 || Size == 16)
11188}
11189
11190static Intrinsic::ID
11192 AtomicRMWInst::BinOp BinOp) {
11193 if (GRLen == 64) {
11194 switch (BinOp) {
11195 default:
11196 llvm_unreachable("Unexpected AtomicRMW BinOp");
11198 return Intrinsic::loongarch_masked_atomicrmw_xchg_i64;
11199 case AtomicRMWInst::Add:
11200 return Intrinsic::loongarch_masked_atomicrmw_add_i64;
11201 case AtomicRMWInst::Sub:
11202 return Intrinsic::loongarch_masked_atomicrmw_sub_i64;
11204 return Intrinsic::loongarch_masked_atomicrmw_nand_i64;
11206 return Intrinsic::loongarch_masked_atomicrmw_umax_i64;
11208 return Intrinsic::loongarch_masked_atomicrmw_umin_i64;
11209 case AtomicRMWInst::Max:
11210 return Intrinsic::loongarch_masked_atomicrmw_max_i64;
11211 case AtomicRMWInst::Min:
11212 return Intrinsic::loongarch_masked_atomicrmw_min_i64;
11213 // TODO: support other AtomicRMWInst.
11214 }
11215 }
11216
11217 if (GRLen == 32) {
11218 switch (BinOp) {
11219 default:
11220 llvm_unreachable("Unexpected AtomicRMW BinOp");
11222 return Intrinsic::loongarch_masked_atomicrmw_xchg_i32;
11223 case AtomicRMWInst::Add:
11224 return Intrinsic::loongarch_masked_atomicrmw_add_i32;
11225 case AtomicRMWInst::Sub:
11226 return Intrinsic::loongarch_masked_atomicrmw_sub_i32;
11228 return Intrinsic::loongarch_masked_atomicrmw_nand_i32;
11230 return Intrinsic::loongarch_masked_atomicrmw_umax_i32;
11232 return Intrinsic::loongarch_masked_atomicrmw_umin_i32;
11233 case AtomicRMWInst::Max:
11234 return Intrinsic::loongarch_masked_atomicrmw_max_i32;
11235 case AtomicRMWInst::Min:
11236 return Intrinsic::loongarch_masked_atomicrmw_min_i32;
11237 // TODO: support other AtomicRMWInst.
11238 }
11239 }
11240
11241 llvm_unreachable("Unexpected GRLen\n");
11242}
11243
11246 const AtomicCmpXchgInst *CI) const {
11247
11248 if (Subtarget.hasLAMCAS())
11250
11252 if (Size == 8 || Size == 16)
11255}
11256
11258 IRBuilderBase &Builder, AtomicCmpXchgInst *CI, Value *AlignedAddr,
11259 Value *CmpVal, Value *NewVal, Value *Mask, AtomicOrdering Ord) const {
11260 unsigned GRLen = Subtarget.getGRLen();
11261 AtomicOrdering FailOrd = CI->getFailureOrdering();
11262 Value *FailureOrdering =
11263 Builder.getIntN(Subtarget.getGRLen(), static_cast<uint64_t>(FailOrd));
11264 Intrinsic::ID CmpXchgIntrID = Intrinsic::loongarch_masked_cmpxchg_i32;
11265 if (GRLen == 64) {
11266 CmpXchgIntrID = Intrinsic::loongarch_masked_cmpxchg_i64;
11267 CmpVal = Builder.CreateSExt(CmpVal, Builder.getInt64Ty());
11268 NewVal = Builder.CreateSExt(NewVal, Builder.getInt64Ty());
11269 Mask = Builder.CreateSExt(Mask, Builder.getInt64Ty());
11270 }
11271 Type *Tys[] = {AlignedAddr->getType()};
11272 Value *Result = Builder.CreateIntrinsic(
11273 CmpXchgIntrID, Tys, {AlignedAddr, CmpVal, NewVal, Mask, FailureOrdering});
11274 if (GRLen == 64)
11275 Result = Builder.CreateTrunc(Result, Builder.getInt32Ty());
11276 return Result;
11277}
11278
11280 IRBuilderBase &Builder, AtomicRMWInst *AI, Value *AlignedAddr, Value *Incr,
11281 Value *Mask, Value *ShiftAmt, AtomicOrdering Ord) const {
11282 // In the case of an atomicrmw xchg with a constant 0/-1 operand, replace
11283 // the atomic instruction with an AtomicRMWInst::And/Or with appropriate
11284 // mask, as this produces better code than the LL/SC loop emitted by
11285 // int_loongarch_masked_atomicrmw_xchg.
11286 if (AI->getOperation() == AtomicRMWInst::Xchg &&
11289 if (CVal->isZero())
11290 return Builder.CreateAtomicRMW(AtomicRMWInst::And, AlignedAddr,
11291 Builder.CreateNot(Mask, "Inv_Mask"),
11292 AI->getAlign(), Ord);
11293 if (CVal->isMinusOne())
11294 return Builder.CreateAtomicRMW(AtomicRMWInst::Or, AlignedAddr, Mask,
11295 AI->getAlign(), Ord);
11296 }
11297
11298 unsigned GRLen = Subtarget.getGRLen();
11299 Value *Ordering =
11300 Builder.getIntN(GRLen, static_cast<uint64_t>(AI->getOrdering()));
11301 Type *Tys[] = {AlignedAddr->getType()};
11303 AI->getModule(),
11305
11306 if (GRLen == 64) {
11307 Incr = Builder.CreateSExt(Incr, Builder.getInt64Ty());
11308 Mask = Builder.CreateSExt(Mask, Builder.getInt64Ty());
11309 ShiftAmt = Builder.CreateSExt(ShiftAmt, Builder.getInt64Ty());
11310 }
11311
11312 Value *Result;
11313
11314 // Must pass the shift amount needed to sign extend the loaded value prior
11315 // to performing a signed comparison for min/max. ShiftAmt is the number of
11316 // bits to shift the value into position. Pass GRLen-ShiftAmt-ValWidth, which
11317 // is the number of bits to left+right shift the value in order to
11318 // sign-extend.
11319 if (AI->getOperation() == AtomicRMWInst::Min ||
11321 const DataLayout &DL = AI->getDataLayout();
11322 unsigned ValWidth =
11323 DL.getTypeStoreSizeInBits(AI->getValOperand()->getType());
11324 Value *SextShamt =
11325 Builder.CreateSub(Builder.getIntN(GRLen, GRLen - ValWidth), ShiftAmt);
11326 Result = Builder.CreateCall(LlwOpScwLoop,
11327 {AlignedAddr, Incr, Mask, SextShamt, Ordering});
11328 } else {
11329 Result =
11330 Builder.CreateCall(LlwOpScwLoop, {AlignedAddr, Incr, Mask, Ordering});
11331 }
11332
11333 if (GRLen == 64)
11334 Result = Builder.CreateTrunc(Result, Builder.getInt32Ty());
11335 return Result;
11336}
11337
11339 const MachineFunction &MF, EVT VT) const {
11340 VT = VT.getScalarType();
11341
11342 if (!VT.isSimple())
11343 return false;
11344
11345 switch (VT.getSimpleVT().SimpleTy) {
11346 case MVT::f32:
11347 case MVT::f64:
11348 return true;
11349 default:
11350 break;
11351 }
11352
11353 return false;
11354}
11355
11357 ExceptionHandling EH, const Constant *PersonalityFn) const {
11358 return LoongArch::R4;
11359}
11360
11362 ExceptionHandling EH, const Constant *PersonalityFn) const {
11363 return LoongArch::R5;
11364}
11365
11366//===----------------------------------------------------------------------===//
11367// Target Optimization Hooks
11368//===----------------------------------------------------------------------===//
11369
11371 const LoongArchSubtarget &Subtarget) {
11372 // Feature FRECIPE instrucions relative accuracy is 2^-14.
11373 // IEEE float has 23 digits and double has 52 digits.
11374 int RefinementSteps = VT.getScalarType() == MVT::f64 ? 2 : 1;
11375 return RefinementSteps;
11376}
11377
11378static bool
11380 assert(Subtarget.hasFrecipe() &&
11381 "Reciprocal estimate queried on unsupported target");
11382
11383 if (!VT.isSimple())
11384 return false;
11385
11386 switch (VT.getSimpleVT().SimpleTy) {
11387 case MVT::f32:
11388 // f32 is the base type for reciprocal estimate instructions.
11389 return true;
11390
11391 case MVT::f64:
11392 return Subtarget.hasBasicD();
11393
11394 case MVT::v4f32:
11395 case MVT::v2f64:
11396 return Subtarget.hasExtLSX();
11397
11398 case MVT::v8f32:
11399 case MVT::v4f64:
11400 return Subtarget.hasExtLASX();
11401
11402 default:
11403 return false;
11404 }
11405}
11406
11408 SelectionDAG &DAG, int Enabled,
11409 int &RefinementSteps,
11410 bool &UseOneConstNR,
11411 bool Reciprocal) const {
11413 "Enabled should never be Disabled here");
11414
11415 if (!Subtarget.hasFrecipe())
11416 return SDValue();
11417
11418 SDLoc DL(Operand);
11419 EVT VT = Operand.getValueType();
11420
11421 // Check supported types.
11422 if (!isSupportedReciprocalEstimateType(VT, Subtarget))
11423 return SDValue();
11424
11425 // Handle refinement steps.
11426 if (RefinementSteps == ReciprocalEstimate::Unspecified)
11427 RefinementSteps = getEstimateRefinementSteps(VT, Subtarget);
11428
11429 // LoongArch only has FRSQRTE which is 1.0 / sqrt(x).
11430 UseOneConstNR = false;
11431 SDValue Rsqrt = DAG.getNode(LoongArchISD::FRSQRTE, DL, VT, Operand);
11432
11433 // If the caller wants 1.0 / sqrt(x), or if further refinement steps
11434 // are needed (which rely on the reciprocal form), return the raw reciprocal
11435 // estimate.
11436 if (Reciprocal || RefinementSteps > 0)
11437 return Rsqrt;
11438
11439 // Otherwise, return sqrt(x) by multiplying with the operand.
11440 return DAG.getNode(ISD::FMUL, DL, VT, Operand, Rsqrt);
11441}
11442
11444 SelectionDAG &DAG,
11445 int Enabled,
11446 int &RefinementSteps) const {
11448 "Enabled should never be Disabled here");
11449
11450 if (!Subtarget.hasFrecipe())
11451 return SDValue();
11452
11453 SDLoc DL(Operand);
11454 EVT VT = Operand.getValueType();
11455
11456 // Check supported types.
11457 if (!isSupportedReciprocalEstimateType(VT, Subtarget))
11458 return SDValue();
11459
11460 if (RefinementSteps == ReciprocalEstimate::Unspecified)
11461 RefinementSteps = getEstimateRefinementSteps(VT, Subtarget);
11462
11463 // FRECIPE computes 1.0 / x.
11464 return DAG.getNode(LoongArchISD::FRECIPE, DL, VT, Operand);
11465}
11466
11467//===----------------------------------------------------------------------===//
11468// LoongArch Inline Assembly Support
11469//===----------------------------------------------------------------------===//
11470
11472LoongArchTargetLowering::getConstraintType(StringRef Constraint) const {
11473 // LoongArch specific constraints in GCC: config/loongarch/constraints.md
11474 //
11475 // 'f': A floating-point register (if available).
11476 // 'k': A memory operand whose address is formed by a base register and
11477 // (optionally scaled) index register.
11478 // 'l': A signed 16-bit constant.
11479 // 'm': A memory operand whose address is formed by a base register and
11480 // offset that is suitable for use in instructions with the same
11481 // addressing mode as st.w and ld.w.
11482 // 'q': A general-purpose register except for $r0 and $r1 (for the csrxchg
11483 // instruction)
11484 // 'I': A signed 12-bit constant (for arithmetic instructions).
11485 // 'J': Integer zero.
11486 // 'K': An unsigned 12-bit constant (for logic instructions).
11487 // "ZB": An address that is held in a general-purpose register. The offset is
11488 // zero.
11489 // "ZC": A memory operand whose address is formed by a base register and
11490 // offset that is suitable for use in instructions with the same
11491 // addressing mode as ll.w and sc.w.
11492 if (Constraint.size() == 1) {
11493 switch (Constraint[0]) {
11494 default:
11495 break;
11496 case 'f':
11497 case 'q':
11498 return C_RegisterClass;
11499 case 'l':
11500 case 'I':
11501 case 'J':
11502 case 'K':
11503 return C_Immediate;
11504 case 'k':
11505 return C_Memory;
11506 }
11507 }
11508
11509 if (Constraint == "ZC" || Constraint == "ZB")
11510 return C_Memory;
11511
11512 // 'm' is handled here.
11513 return TargetLowering::getConstraintType(Constraint);
11514}
11515
11516InlineAsm::ConstraintCode LoongArchTargetLowering::getInlineAsmMemConstraint(
11517 StringRef ConstraintCode) const {
11518 return StringSwitch<InlineAsm::ConstraintCode>(ConstraintCode)
11522 .Default(TargetLowering::getInlineAsmMemConstraint(ConstraintCode));
11523}
11524
11525std::pair<unsigned, const TargetRegisterClass *>
11526LoongArchTargetLowering::getRegForInlineAsmConstraint(
11527 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
11528 // First, see if this is a constraint that directly corresponds to a LoongArch
11529 // register class.
11530 if (Constraint.size() == 1) {
11531 switch (Constraint[0]) {
11532 case 'r':
11533 // TODO: Support fixed vectors up to GRLen?
11534 if (VT.isVector())
11535 break;
11536 return std::make_pair(0U, &LoongArch::GPRRegClass);
11537 case 'q':
11538 return std::make_pair(0U, &LoongArch::GPRNoR0R1RegClass);
11539 case 'f':
11540 if (Subtarget.hasBasicF() && VT == MVT::f32)
11541 return std::make_pair(0U, &LoongArch::FPR32RegClass);
11542 if (Subtarget.hasBasicD() && VT == MVT::f64)
11543 return std::make_pair(0U, &LoongArch::FPR64RegClass);
11544 if (Subtarget.hasExtLSX() &&
11545 TRI->isTypeLegalForClass(LoongArch::LSX128RegClass, VT))
11546 return std::make_pair(0U, &LoongArch::LSX128RegClass);
11547 if (Subtarget.hasExtLSX() && VT == MVT::i128)
11548 return std::make_pair(0U, &LoongArch::LSX128RegClass);
11549 if (Subtarget.hasExtLASX() &&
11550 TRI->isTypeLegalForClass(LoongArch::LASX256RegClass, VT))
11551 return std::make_pair(0U, &LoongArch::LASX256RegClass);
11552 break;
11553 default:
11554 break;
11555 }
11556 }
11557
11558 // TargetLowering::getRegForInlineAsmConstraint uses the name of the TableGen
11559 // record (e.g. the "R0" in `def R0`) to choose registers for InlineAsm
11560 // constraints while the official register name is prefixed with a '$'. So we
11561 // clip the '$' from the original constraint string (e.g. {$r0} to {r0}.)
11562 // before it being parsed. And TargetLowering::getRegForInlineAsmConstraint is
11563 // case insensitive, so no need to convert the constraint to upper case here.
11564 //
11565 // For now, no need to support ABI names (e.g. `$a0`) as clang will correctly
11566 // decode the usage of register name aliases into their official names. And
11567 // AFAIK, the not yet upstreamed `rustc` for LoongArch will always use
11568 // official register names.
11569 if (Constraint.starts_with("{$r") || Constraint.starts_with("{$f") ||
11570 Constraint.starts_with("{$vr") || Constraint.starts_with("{$xr")) {
11571 bool IsFP = Constraint[2] == 'f';
11572 std::pair<StringRef, StringRef> Temp = Constraint.split('$');
11573 std::pair<unsigned, const TargetRegisterClass *> R;
11575 TRI, join_items("", Temp.first, Temp.second), VT);
11576 // Match those names to the widest floating point register type available.
11577 if (IsFP) {
11578 unsigned RegNo = R.first;
11579 if (LoongArch::F0 <= RegNo && RegNo <= LoongArch::F31) {
11580 if (Subtarget.hasBasicD() && (VT == MVT::f64 || VT == MVT::Other)) {
11581 unsigned DReg = RegNo - LoongArch::F0 + LoongArch::F0_64;
11582 return std::make_pair(DReg, &LoongArch::FPR64RegClass);
11583 }
11584 }
11585 }
11586 return R;
11587 }
11588
11589 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
11590}
11591
11592void LoongArchTargetLowering::LowerAsmOperandForConstraint(
11593 SDValue Op, StringRef Constraint, std::vector<SDValue> &Ops,
11594 SelectionDAG &DAG) const {
11595 // Currently only support length 1 constraints.
11596 if (Constraint.size() == 1) {
11597 switch (Constraint[0]) {
11598 case 'l':
11599 // Validate & create a 16-bit signed immediate operand.
11600 if (auto *C = dyn_cast<ConstantSDNode>(Op)) {
11601 uint64_t CVal = C->getSExtValue();
11602 if (isInt<16>(CVal))
11603 Ops.push_back(DAG.getSignedTargetConstant(CVal, SDLoc(Op),
11604 Subtarget.getGRLenVT()));
11605 }
11606 return;
11607 case 'I':
11608 // Validate & create a 12-bit signed immediate operand.
11609 if (auto *C = dyn_cast<ConstantSDNode>(Op)) {
11610 uint64_t CVal = C->getSExtValue();
11611 if (isInt<12>(CVal))
11612 Ops.push_back(DAG.getSignedTargetConstant(CVal, SDLoc(Op),
11613 Subtarget.getGRLenVT()));
11614 }
11615 return;
11616 case 'J':
11617 // Validate & create an integer zero operand.
11618 if (auto *C = dyn_cast<ConstantSDNode>(Op))
11619 if (C->getZExtValue() == 0)
11620 Ops.push_back(
11621 DAG.getTargetConstant(0, SDLoc(Op), Subtarget.getGRLenVT()));
11622 return;
11623 case 'K':
11624 // Validate & create a 12-bit unsigned immediate operand.
11625 if (auto *C = dyn_cast<ConstantSDNode>(Op)) {
11626 uint64_t CVal = C->getZExtValue();
11627 if (isUInt<12>(CVal))
11628 Ops.push_back(
11629 DAG.getTargetConstant(CVal, SDLoc(Op), Subtarget.getGRLenVT()));
11630 }
11631 return;
11632 default:
11633 break;
11634 }
11635 }
11637}
11638
11639#define GET_REGISTER_MATCHER
11640#include "LoongArchGenAsmMatcher.inc"
11641
11644 const MachineFunction &MF) const {
11645 std::pair<StringRef, StringRef> Name = StringRef(RegName).split('$');
11646 std::string NewRegName = Name.second.str();
11647 Register Reg = MatchRegisterAltName(NewRegName);
11648 if (!Reg)
11649 Reg = MatchRegisterName(NewRegName);
11650 if (!Reg)
11651 return Reg;
11652 BitVector ReservedRegs = Subtarget.getRegisterInfo()->getReservedRegs(MF);
11653 if (!ReservedRegs.test(Reg))
11654 report_fatal_error(Twine("Trying to obtain non-reserved register \"" +
11655 StringRef(RegName) + "\"."));
11656 return Reg;
11657}
11658
11660 EVT VT, SDValue C) const {
11661 // TODO: Support vectors.
11662 if (!VT.isScalarInteger())
11663 return false;
11664
11665 // Omit the optimization if the data size exceeds GRLen.
11666 if (VT.getSizeInBits() > Subtarget.getGRLen())
11667 return false;
11668
11669 if (auto *ConstNode = dyn_cast<ConstantSDNode>(C.getNode())) {
11670 const APInt &Imm = ConstNode->getAPIntValue();
11671 // Break MUL into (SLLI + ADD/SUB) or ALSL.
11672 if ((Imm + 1).isPowerOf2() || (Imm - 1).isPowerOf2() ||
11673 (1 - Imm).isPowerOf2() || (-1 - Imm).isPowerOf2())
11674 return true;
11675 // Break MUL into (ALSL x, (SLLI x, imm0), imm1).
11676 if (ConstNode->hasOneUse() &&
11677 ((Imm - 2).isPowerOf2() || (Imm - 4).isPowerOf2() ||
11678 (Imm - 8).isPowerOf2() || (Imm - 16).isPowerOf2()))
11679 return true;
11680 // Break (MUL x, imm) into (ADD (SLLI x, s0), (SLLI x, s1)),
11681 // in which the immediate has two set bits. Or Break (MUL x, imm)
11682 // into (SUB (SLLI x, s0), (SLLI x, s1)), in which the immediate
11683 // equals to (1 << s0) - (1 << s1).
11684 if (ConstNode->hasOneUse() && !(Imm.sge(-2048) && Imm.sle(4095))) {
11685 unsigned Shifts = Imm.countr_zero();
11686 // Reject immediates which can be composed via a single LUI.
11687 if (Shifts >= 12)
11688 return false;
11689 // Reject multiplications can be optimized to
11690 // (SLLI (ALSL x, x, 1/2/3/4), s).
11691 APInt ImmPop = Imm.ashr(Shifts);
11692 if (ImmPop == 3 || ImmPop == 5 || ImmPop == 9 || ImmPop == 17)
11693 return false;
11694 // We do not consider the case `(-Imm - ImmSmall).isPowerOf2()`,
11695 // since it needs one more instruction than other 3 cases.
11696 APInt ImmSmall = APInt(Imm.getBitWidth(), 1ULL << Shifts, true);
11697 if ((Imm - ImmSmall).isPowerOf2() || (Imm + ImmSmall).isPowerOf2() ||
11698 (ImmSmall - Imm).isPowerOf2())
11699 return true;
11700 }
11701 }
11702
11703 return false;
11704}
11705
11707 const AddrMode &AM,
11708 Type *Ty, unsigned AS,
11709 Instruction *I) const {
11710 // LoongArch has four basic addressing modes:
11711 // 1. reg
11712 // 2. reg + 12-bit signed offset
11713 // 3. reg + 14-bit signed offset left-shifted by 2
11714 // 4. reg1 + reg2
11715 // TODO: Add more checks after support vector extension.
11716
11717 // No global is ever allowed as a base.
11718 if (AM.BaseGV)
11719 return false;
11720
11721 // Require a 12-bit signed offset or 14-bit signed offset left-shifted by 2
11722 // with `UAL` feature.
11723 if (!isInt<12>(AM.BaseOffs) &&
11724 !(isShiftedInt<14, 2>(AM.BaseOffs) && Subtarget.hasUAL()))
11725 return false;
11726
11727 switch (AM.Scale) {
11728 case 0:
11729 // "r+i" or just "i", depending on HasBaseReg.
11730 break;
11731 case 1:
11732 // "r+r+i" is not allowed.
11733 if (AM.HasBaseReg && AM.BaseOffs)
11734 return false;
11735 // Otherwise we have "r+r" or "r+i".
11736 break;
11737 case 2:
11738 // "2*r+r" or "2*r+i" is not allowed.
11739 if (AM.HasBaseReg || AM.BaseOffs)
11740 return false;
11741 // Allow "2*r" as "r+r".
11742 break;
11743 default:
11744 return false;
11745 }
11746
11747 return true;
11748}
11749
11751 return isInt<12>(Imm);
11752}
11753
11755 return isInt<12>(Imm);
11756}
11757
11759 // Zexts are free if they can be combined with a load.
11760 // Don't advertise i32->i64 zextload as being free for LA64. It interacts
11761 // poorly with type legalization of compares preferring sext.
11762 if (auto *LD = dyn_cast<LoadSDNode>(Val)) {
11763 EVT MemVT = LD->getMemoryVT();
11764 if ((MemVT == MVT::i8 || MemVT == MVT::i16) &&
11765 (LD->getExtensionType() == ISD::NON_EXTLOAD ||
11766 LD->getExtensionType() == ISD::ZEXTLOAD))
11767 return true;
11768 }
11769
11770 return TargetLowering::isZExtFree(Val, VT2);
11771}
11772
11774 EVT DstVT) const {
11775 return Subtarget.is64Bit() && SrcVT == MVT::i32 && DstVT == MVT::i64;
11776}
11777
11779 return Subtarget.is64Bit() && CI->getType()->isIntegerTy(32);
11780}
11781
11783 // TODO: Support vectors.
11784 if (Y.getValueType().isVector())
11785 return false;
11786
11787 return !isa<ConstantSDNode>(Y);
11788}
11789
11791 // LAMCAS will use amcas[_DB].{b/h/w/d} which does not require extension.
11792 return Subtarget.hasLAMCAS() ? ISD::ANY_EXTEND : ISD::SIGN_EXTEND;
11793}
11794
11796 Type *Ty, bool IsSigned) const {
11797 if (Subtarget.is64Bit() && Ty->isIntegerTy(32))
11798 return true;
11799
11800 return IsSigned;
11801}
11802
11804 // Return false to suppress the unnecessary extensions if the LibCall
11805 // arguments or return value is a float narrower than GRLEN on a soft FP ABI.
11806 if (Subtarget.isSoftFPABI() && (Type.isFloatingPoint() && !Type.isVector() &&
11807 Type.getSizeInBits() < Subtarget.getGRLen()))
11808 return false;
11809 return true;
11810}
11811
11812// memcpy, and other memory intrinsics, typically tries to use wider load/store
11813// if the source/dest is aligned and the copy size is large enough. We therefore
11814// want to align such objects passed to memory intrinsics.
11816 unsigned &MinSize,
11817 Align &PrefAlign) const {
11818 if (!isa<MemIntrinsic>(CI))
11819 return false;
11820
11821 if (Subtarget.is64Bit()) {
11822 MinSize = 8;
11823 PrefAlign = Align(8);
11824 } else {
11825 MinSize = 4;
11826 PrefAlign = Align(4);
11827 }
11828
11829 return true;
11830}
11831
11834 if (!VT.isScalableVector() && VT.getVectorNumElements() != 1 &&
11835 VT.getVectorElementType() != MVT::i1)
11836 return TypeWidenVector;
11837
11839}
11840
11841bool LoongArchTargetLowering::splitValueIntoRegisterParts(
11842 SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts,
11843 unsigned NumParts, MVT PartVT, std::optional<CallingConv::ID> CC) const {
11844 bool IsABIRegCopy = CC.has_value();
11845 EVT ValueVT = Val.getValueType();
11846
11847 if (IsABIRegCopy && (ValueVT == MVT::f16 || ValueVT == MVT::bf16) &&
11848 PartVT == MVT::f32) {
11849 // Cast the [b]f16 to i16, extend to i32, pad with ones to make a float
11850 // nan, and cast to f32.
11851 Val = DAG.getNode(ISD::BITCAST, DL, MVT::i16, Val);
11852 Val = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, Val);
11853 Val = DAG.getNode(ISD::OR, DL, MVT::i32, Val,
11854 DAG.getConstant(0xFFFF0000, DL, MVT::i32));
11855 Val = DAG.getNode(ISD::BITCAST, DL, MVT::f32, Val);
11856 Parts[0] = Val;
11857 return true;
11858 }
11859
11860 return false;
11861}
11862
11863SDValue LoongArchTargetLowering::joinRegisterPartsIntoValue(
11864 SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts,
11865 MVT PartVT, EVT ValueVT, std::optional<CallingConv::ID> CC) const {
11866 bool IsABIRegCopy = CC.has_value();
11867
11868 if (IsABIRegCopy && (ValueVT == MVT::f16 || ValueVT == MVT::bf16) &&
11869 PartVT == MVT::f32) {
11870 SDValue Val = Parts[0];
11871
11872 // Cast the f32 to i32, truncate to i16, and cast back to [b]f16.
11873 Val = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Val);
11874 Val = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, Val);
11875 Val = DAG.getNode(ISD::BITCAST, DL, ValueVT, Val);
11876 return Val;
11877 }
11878
11879 return SDValue();
11880}
11881
11882MVT LoongArchTargetLowering::getRegisterTypeForCallingConv(LLVMContext &Context,
11883 CallingConv::ID CC,
11884 EVT VT) const {
11885 // Use f32 to pass f16.
11886 if (VT == MVT::f16 && Subtarget.hasBasicF())
11887 return MVT::f32;
11888
11890}
11891
11892unsigned LoongArchTargetLowering::getNumRegistersForCallingConv(
11893 LLVMContext &Context, CallingConv::ID CC, EVT VT) const {
11894 // Use f32 to pass f16.
11895 if (VT == MVT::f16 && Subtarget.hasBasicF())
11896 return 1;
11897
11899}
11900
11902 const SDValue Op, KnownBits &Known, const APInt &DemandedElts,
11903 const SelectionDAG &DAG, unsigned Depth) const {
11904 unsigned Opc = Op.getOpcode();
11905 Known.resetAll();
11906 switch (Opc) {
11907 default:
11908 break;
11909 case LoongArchISD::VANYNONZERO:
11910 case LoongArchISD::VALLZERO: {
11911 // MOVCF2GR zero-extend the i1 cond to GPR.
11912 Known.Zero.setBitsFrom(1);
11913 break;
11914 }
11915 case LoongArchISD::VPICK_ZEXT_ELT: {
11916 assert(isa<VTSDNode>(Op->getOperand(2)) && "Unexpected operand!");
11917 EVT VT = cast<VTSDNode>(Op->getOperand(2))->getVT();
11918 unsigned VTBits = VT.getScalarSizeInBits();
11919 assert(Known.getBitWidth() >= VTBits && "Unexpected width!");
11920 Known.Zero.setBitsFrom(VTBits);
11921 break;
11922 }
11923 }
11924}
11925
11927 SDValue Op, const APInt &OriginalDemandedBits,
11928 const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO,
11929 unsigned Depth) const {
11930 EVT VT = Op.getValueType();
11931 unsigned BitWidth = OriginalDemandedBits.getBitWidth();
11932 unsigned Opc = Op.getOpcode();
11933 switch (Opc) {
11934 default:
11935 break;
11936 case LoongArchISD::CRC_W_B_W:
11937 case LoongArchISD::CRC_W_H_W:
11938 case LoongArchISD::CRCC_W_B_W:
11939 case LoongArchISD::CRCC_W_H_W: {
11940 KnownBits KnownSrc;
11941 APInt DemandedSrcBits =
11942 APInt::getLowBitsSet(BitWidth, (Opc == LoongArchISD::CRC_W_B_W ||
11943 Opc == LoongArchISD::CRCC_W_B_W)
11944 ? 8
11945 : 16);
11946 return SimplifyDemandedBits(Op.getOperand(1), DemandedSrcBits,
11947 OriginalDemandedElts, KnownSrc, TLO, Depth + 1);
11948 }
11949 case LoongArchISD::VMSKLTZ:
11950 case LoongArchISD::XVMSKLTZ: {
11951 SDValue Src = Op.getOperand(0);
11952 MVT SrcVT = Src.getSimpleValueType();
11953 unsigned SrcBits = SrcVT.getScalarSizeInBits();
11954 unsigned NumElts = SrcVT.getVectorNumElements();
11955
11956 // If we don't need the sign bits at all just return zero.
11957 if (OriginalDemandedBits.countr_zero() >= NumElts)
11958 return TLO.CombineTo(Op, TLO.DAG.getConstant(0, SDLoc(Op), VT));
11959
11960 // Only demand the vector elements of the sign bits we need.
11961 APInt KnownUndef, KnownZero;
11962 APInt DemandedElts = OriginalDemandedBits.zextOrTrunc(NumElts);
11963 if (SimplifyDemandedVectorElts(Src, DemandedElts, KnownUndef, KnownZero,
11964 TLO, Depth + 1))
11965 return true;
11966
11967 Known.Zero = KnownZero.zext(BitWidth);
11968 Known.Zero.setHighBits(BitWidth - NumElts);
11969
11970 // [X]VMSKLTZ only uses the MSB from each vector element.
11971 KnownBits KnownSrc;
11972 APInt DemandedSrcBits = APInt::getSignMask(SrcBits);
11973 if (SimplifyDemandedBits(Src, DemandedSrcBits, DemandedElts, KnownSrc, TLO,
11974 Depth + 1))
11975 return true;
11976
11977 if (KnownSrc.One[SrcBits - 1])
11978 Known.One.setLowBits(NumElts);
11979 else if (KnownSrc.Zero[SrcBits - 1])
11980 Known.Zero.setLowBits(NumElts);
11981
11982 // Attempt to avoid multi-use ops if we don't need anything from it.
11984 Src, DemandedSrcBits, DemandedElts, TLO.DAG, Depth + 1))
11985 return TLO.CombineTo(Op, TLO.DAG.getNode(Opc, SDLoc(Op), VT, NewSrc));
11986 return false;
11987 }
11988 }
11989
11991 Op, OriginalDemandedBits, OriginalDemandedElts, Known, TLO, Depth);
11992}
11993
11995 unsigned Opc = VecOp.getOpcode();
11996
11997 // Assume target opcodes can't be scalarized.
11998 // TODO - do we have any exceptions?
11999 if (Opc >= ISD::BUILTIN_OP_END || !isBinOp(Opc))
12000 return false;
12001
12002 // If the vector op is not supported, try to convert to scalar.
12003 EVT VecVT = VecOp.getValueType();
12005 return true;
12006
12007 // If the vector op is supported, but the scalar op is not, the transform may
12008 // not be worthwhile.
12009 EVT ScalarVT = VecVT.getScalarType();
12010 return isOperationLegalOrCustomOrPromote(Opc, ScalarVT);
12011}
12012
12015 unsigned Index) const {
12018
12019 // Extract a 128-bit subvector from index 0 of a 256-bit vector is free.
12020 if (Index == 0)
12023}
12024
12026 unsigned Index) const {
12027 EVT EltVT = VT.getScalarType();
12028
12029 // Extract a scalar FP value from index 0 of a vector is free.
12030 return (EltVT == MVT::f32 || EltVT == MVT::f64) && Index == 0;
12031}
12032
12034 const MachineFunction &MF) const {
12035
12036 // If the function specifically requests inline stack probes, emit them.
12037 if (MF.getFunction().hasFnAttribute("probe-stack"))
12038 return MF.getFunction().getFnAttribute("probe-stack").getValueAsString() ==
12039 "inline-asm";
12040
12041 return false;
12042}
12043
12045 Align StackAlign) const {
12046 // The default stack probe size is 4096 if the function has no
12047 // stack-probe-size attribute.
12048 const Function &Fn = MF.getFunction();
12049 unsigned StackProbeSize =
12050 Fn.getFnAttributeAsParsedInteger("stack-probe-size", 4096);
12051 // Round down to the stack alignment.
12052 StackProbeSize = alignDown(StackProbeSize, StackAlign.value());
12053 return StackProbeSize ? StackProbeSize : StackAlign.value();
12054}
12055
12056SDValue
12057LoongArchTargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op,
12058 SelectionDAG &DAG) const {
12060 if (!hasInlineStackProbe(MF))
12061 return SDValue();
12062
12063 const MVT GRLenVT = Subtarget.getGRLenVT();
12064 // Get the inputs.
12065 SDValue Chain = Op.getOperand(0);
12066 SDValue Size = Op.getOperand(1);
12067
12068 const MaybeAlign Align =
12069 cast<ConstantSDNode>(Op.getOperand(2))->getMaybeAlignValue();
12070 const SDLoc dl(Op);
12071 const EVT VT = Op.getValueType();
12072
12073 // Construct the new SP value in a GPR.
12074 SDValue SP = DAG.getCopyFromReg(Chain, dl, LoongArch::R3, GRLenVT);
12075 Chain = SP.getValue(1);
12076 SP = DAG.getNode(ISD::SUB, dl, GRLenVT, SP, Size);
12077 if (Align)
12078 SP = DAG.getNode(ISD::AND, dl, VT, SP.getValue(0),
12079 DAG.getSignedConstant(-Align->value(), dl, VT));
12080
12081 // Set the real SP to the new value with a probing loop.
12082 Chain = DAG.getNode(LoongArchISD::PROBED_ALLOCA, dl, MVT::Other, Chain, SP);
12083 return DAG.getMergeValues({SP, Chain}, dl);
12084}
12085
12088 MachineBasicBlock *MBB) const {
12089 MachineFunction &MF = *MBB->getParent();
12090 MachineBasicBlock::iterator MBBI = MI.getIterator();
12091 DebugLoc DL = MBB->findDebugLoc(MBBI);
12092 const Register TargetReg = MI.getOperand(0).getReg();
12093
12094 const LoongArchInstrInfo *TII = Subtarget.getInstrInfo();
12095 const bool IsLA64 = Subtarget.is64Bit();
12096 const Align StackAlign = Subtarget.getFrameLowering()->getStackAlign();
12097 const LoongArchTargetLowering *TLI = Subtarget.getTargetLowering();
12098 const uint64_t ProbeSize = TLI->getStackProbeSize(MF, StackAlign);
12099
12100 MachineFunction::iterator MBBInsertPoint = std::next(MBB->getIterator());
12101 MachineBasicBlock *const LoopTestMBB =
12102 MF.CreateMachineBasicBlock(MBB->getBasicBlock());
12103 MF.insert(MBBInsertPoint, LoopTestMBB);
12104 MachineBasicBlock *const ExitMBB =
12105 MF.CreateMachineBasicBlock(MBB->getBasicBlock());
12106 MF.insert(MBBInsertPoint, ExitMBB);
12107 const Register SPReg = LoongArch::R3;
12108 const Register ScratchReg =
12109 MF.getRegInfo().createVirtualRegister(&LoongArch::GPRRegClass);
12110
12111 // ScratchReg = ProbeSize
12112 TII->movImm(*MBB, MBBI, DL, ScratchReg, ProbeSize, MachineInstr::NoFlags);
12113
12114 // LoopTest:
12115 // sub.{w/d} $sp, $sp, ScratchReg
12116 BuildMI(*LoopTestMBB, LoopTestMBB->end(), DL,
12117 TII->get(IsLA64 ? LoongArch::SUB_D : LoongArch::SUB_W), SPReg)
12118 .addReg(SPReg)
12119 .addReg(ScratchReg);
12120
12121 // st.{w/d} $zero, $sp, 0
12122 BuildMI(*LoopTestMBB, LoopTestMBB->end(), DL,
12123 TII->get(IsLA64 ? LoongArch::ST_D : LoongArch::ST_W))
12124 .addReg(LoongArch::R0)
12125 .addReg(SPReg)
12126 .addImm(0);
12127
12128 // bltu TargetReg, $sp, LoopTest
12129 BuildMI(*LoopTestMBB, LoopTestMBB->end(), DL, TII->get(LoongArch::BLTU))
12130 .addReg(TargetReg)
12131 .addReg(SPReg)
12132 .addMBB(LoopTestMBB);
12133
12134 // move $sp, TargetReg
12135 BuildMI(*ExitMBB, ExitMBB->end(), DL, TII->get(LoongArch::OR), SPReg)
12136 .addReg(TargetReg)
12137 .addReg(LoongArch::R0);
12138
12139 ExitMBB->splice(ExitMBB->end(), MBB, std::next(MBBI), MBB->end());
12141
12142 LoopTestMBB->addSuccessor(ExitMBB);
12143 LoopTestMBB->addSuccessor(LoopTestMBB);
12144 MBB->addSuccessor(LoopTestMBB);
12145
12146 MI.eraseFromParent();
12147 MF.getInfo<LoongArchMachineFunctionInfo>()->setDynamicAllocation();
12148 return ExitMBB->begin()->getParent();
12149}
static MCRegister MatchRegisterName(StringRef Name)
static SDValue performSHLCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG)
If the operand is a bitwise AND with a constant RHS, and the shift has a constant RHS and is the only...
static bool checkValueWidth(SDValue V, unsigned width, ISD::LoadExtType &ExtType)
static SDValue performORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue performANDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue performSELECT_CCCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG)
static SDValue performSETCCCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, SelectionDAG &DAG)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static MCRegister MatchRegisterAltName(StringRef Name)
Maps from the set of all alternative registernames to a register number.
Function Alias Analysis Results
static uint64_t getConstant(const Value *IndexValue)
static SDValue getTargetNode(ConstantPoolSDNode *N, const SDLoc &DL, EVT Ty, SelectionDAG &DAG, unsigned Flags)
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static SDValue convertValVTToLocVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
static SDValue unpackFromMemLoc(SelectionDAG &DAG, SDValue Chain, const CCValAssign &VA, const SDLoc &DL)
static SDValue convertLocVTToValVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
static MachineBasicBlock * emitSelectPseudo(MachineInstr &MI, MachineBasicBlock *BB, unsigned Opcode)
static SDValue unpackFromRegLoc(const CSKYSubtarget &Subtarget, SelectionDAG &DAG, SDValue Chain, const CCValAssign &VA, const SDLoc &DL)
#define clEnumValN(ENUMVAL, FLAGNAME, DESC)
static bool isSigned(unsigned Opcode)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
static SDValue performINTRINSIC_WO_CHAINCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue performADDCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
const MCPhysReg ArgFPR32s[]
static SDValue lower128BitShuffle(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Dispatching routine to lower various 128-bit LoongArch vector shuffles.
static SDValue lowerVECTOR_SHUFFLE_XVSHUF4I(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVSHUF4I (if possible).
const MCPhysReg ArgVRs[]
static SDValue lowerVECTOR_SHUFFLE_VPICKEV(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VPICKEV (if possible).
static SDValue combineSelectToBinOp(SDNode *N, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_XVPICKOD(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVPICKOD (if possible).
static SDValue unpackF64OnLA32DSoftABI(SelectionDAG &DAG, SDValue Chain, const CCValAssign &VA, const CCValAssign &HiVA, const SDLoc &DL)
static bool fitsRegularPattern(typename SmallVectorImpl< ValType >::const_iterator Begin, unsigned CheckStride, typename SmallVectorImpl< ValType >::const_iterator End, ValType ExpectedIndex, unsigned ExpectedIndexStride)
Determine whether a range fits a regular pattern of values.
static SDValue lowerVECTOR_SHUFFLE_IsReverse(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE whose result is the reversed source vector.
static SDValue PromoteMaskArithmetic(SDValue N, const SDLoc &DL, EVT VT, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget, unsigned Depth)
static SDValue performUINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue performHorizWideningCombine(SDNode *N, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static std::pair< SDValue, EVT > stripSignExtendInReg(SDValue V)
static SDValue emitIntrinsicErrorMessage(SDValue Op, StringRef ErrorMsg, SelectionDAG &DAG)
static SDValue ExtendSrcToDst(SDNode *N, SelectionDAG &DAG, unsigned ExtendOp)
static cl::opt< bool > ZeroDivCheck("loongarch-check-zero-division", cl::Hidden, cl::desc("Trap on integer division by zero."), cl::init(false))
static SDValue lowerVECTOR_SHUFFLE_XVPERMI(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVPERMI (if possible).
static SDValue lowerVECTOR_SHUFFLE_VSHUF(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into VSHUF.
static int getEstimateRefinementSteps(EVT VT, const LoongArchSubtarget &Subtarget)
static bool isSupportedReciprocalEstimateType(EVT VT, const LoongArchSubtarget &Subtarget)
static void emitErrorAndReplaceIntrinsicResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG, StringRef ErrorMsg, bool WithChain=true)
static SDValue lowerVECTOR_SHUFFLEAsByteRotate(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE as byte rotate (if possible).
static SDValue checkIntrinsicImmArg(SDValue Op, unsigned ImmOp, SelectionDAG &DAG, bool IsSigned=false)
static SDValue performSUBCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_XVINSVE0(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVINSVE0 (if possible).
static SDValue performMOVFR2GR_SCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_VILVH(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VILVH (if possible).
static SDValue performDemandedBitsCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI)
static bool CC_LoongArch(const DataLayout &DL, LoongArchABI::ABI ABI, unsigned ValNo, MVT ValVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, CCState &State, bool IsRet, Type *OrigTy)
static std::tuple< unsigned, SDValue, EVT > matchBinOpWithSharedOperand(SDValue BinV, SDValue X)
static SDValue performVSELECTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static Align getPrefTypeAlign(EVT VT, SelectionDAG &DAG)
static SDValue performSPLIT_PAIR_F64Combine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue performBITCASTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue performSRLCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static MachineBasicBlock * emitSplitPairF64Pseudo(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
static SDValue lowerVectorBitSetImm(SDNode *Node, SelectionDAG &DAG)
static SDValue performSETCC_BITCASTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
@ NoMaterializeFPImm
@ MaterializeFPImm2Ins
@ MaterializeFPImm5Ins
@ MaterializeFPImm6Ins
@ MaterializeFPImm3Ins
@ MaterializeFPImm4Ins
static SDValue performEXTENDCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_XVPACKOD(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVPACKOD (if possible).
static bool buildVPERMIInfo(ArrayRef< int > Mask, SDValue V1, SDValue V2, SmallVectorImpl< SDValue > &SrcVec, unsigned &MaskImm)
static std::optional< bool > matchSetCC(SDValue LHS, SDValue RHS, ISD::CondCode CC, SDValue Val)
static SDValue combineAndNotIntoVANDN(SDNode *N, const SDLoc &DL, SelectionDAG &DAG)
Try to fold: (and (xor X, -1), Y) -> (vandn X, Y).
static SDValue lowerBUILD_VECTORAsBroadCastLoad(BuildVectorSDNode *BVOp, const SDLoc &DL, SelectionDAG &DAG)
#define CRC_CASE_EXT_BINARYOP(NAME, NODE)
static SDValue lowerVectorBitRevImm(SDNode *Node, SelectionDAG &DAG)
static bool checkBitcastSrcVectorSize(SDValue Src, unsigned Size, unsigned Depth)
static bool isConstantSplatVector(SDValue N, APInt &SplatValue, unsigned MinSizeInBits)
static SDValue lowerVECTOR_SHUFFLEAsShift(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget, const APInt &Zeroable)
Lower VECTOR_SHUFFLE as shift (if possible).
static SDValue lowerVECTOR_SHUFFLE_VSHUF4I(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into VSHUF4I (if possible).
static SDValue truncateVecElts(SDNode *Node, SelectionDAG &DAG)
static bool CC_LoongArch_GHC(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
static MachineBasicBlock * insertDivByZeroTrap(MachineInstr &MI, MachineBasicBlock *MBB)
static SDValue customLegalizeToWOpWithSExt(SDNode *N, SelectionDAG &DAG)
static SDValue lowerVECTOR_SHUFFLE_VEXTRINS(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into VEXTRINS (if possible).
static SDValue lowerVectorBitClear(SDNode *Node, SelectionDAG &DAG)
static SDValue performSELECTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_VPACKEV(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VPACKEV (if possible).
static MachineBasicBlock * emitPseudoVMSKCOND(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
static SDValue performSINT_TO_FPCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue performVANDNCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
Do target-specific dag combines on LoongArchISD::VANDN nodes.
static void replaceVPICKVE2GRResults(SDNode *Node, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget, unsigned ResOp)
static SDValue lowerVECTOR_SHUFFLEAsZeroOrAnyExtend(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const APInt &Zeroable)
Lower VECTOR_SHUFFLE as ZERO_EXTEND Or ANY_EXTEND (if possible).
static SDValue legalizeIntrinsicImmArg(SDNode *Node, unsigned ImmOp, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget, bool IsSigned=false)
static cl::opt< MaterializeFPImm > MaterializeFPImmInsNum("loongarch-materialize-float-imm", cl::Hidden, cl::desc("Maximum number of instructions used (including code sequence " "to generate the value and moving the value to FPR) when " "materializing floating-point immediates (default = 3)"), cl::init(MaterializeFPImm3Ins), cl::values(clEnumValN(NoMaterializeFPImm, "0", "Use constant pool"), clEnumValN(MaterializeFPImm2Ins, "2", "Materialize FP immediate within 2 instructions"), clEnumValN(MaterializeFPImm3Ins, "3", "Materialize FP immediate within 3 instructions"), clEnumValN(MaterializeFPImm4Ins, "4", "Materialize FP immediate within 4 instructions"), clEnumValN(MaterializeFPImm5Ins, "5", "Materialize FP immediate within 5 instructions"), clEnumValN(MaterializeFPImm6Ins, "6", "Materialize FP immediate within 6 instructions " "(behaves same as 5 on loongarch64)")))
static SDValue emitIntrinsicWithChainErrorMessage(SDValue Op, StringRef ErrorMsg, SelectionDAG &DAG)
const MCPhysReg ArgXRs[]
static bool CC_LoongArchAssign2GRLen(unsigned GRLen, CCState &State, CCValAssign VA1, ISD::ArgFlagsTy ArgFlags1, unsigned ValNo2, MVT ValVT2, MVT LocVT2, ISD::ArgFlagsTy ArgFlags2)
static unsigned getLoongArchWOpcode(unsigned Opcode)
const MCPhysReg ArgFPR64s[]
static MachineBasicBlock * emitPseudoCTPOP(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
static SDValue performMOVGR2FR_WCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
#define IOCSRWR_CASE(NAME, NODE)
#define CRC_CASE_EXT_UNARYOP(NAME, NODE)
static SDValue lowerVECTOR_SHUFFLE_VPACKOD(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VPACKOD (if possible).
static SDValue signExtendBitcastSrcVector(SelectionDAG &DAG, EVT SExtVT, SDValue Src, const SDLoc &DL)
static SDValue isNOT(SDValue V, SelectionDAG &DAG)
static SDValue lower256BitShuffle(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Dispatching routine to lower various 256-bit LoongArch vector shuffles.
static SDValue lowerVECTOR_SHUFFLE_VREPLVEI(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into VREPLVEI (if possible).
static MachineBasicBlock * emitPseudoXVINSGR2VR(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
const MCPhysReg PreserveNoneArgGPRs[]
static void fillVector(ArrayRef< SDValue > Ops, SelectionDAG &DAG, SDLoc DL, const LoongArchSubtarget &Subtarget, SDValue &Vector, EVT ResTy)
static SDValue fillSubVectorFromBuildVector(BuildVectorSDNode *Node, SelectionDAG &DAG, SDLoc DL, const LoongArchSubtarget &Subtarget, EVT ResTy, unsigned first)
static bool isSelectPseudo(MachineInstr &MI)
static SDValue foldVMskZeroTest(SDValue LHS, SDValue RHS, ISD::CondCode CC, const SDLoc &DL, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static SDValue foldBinOpIntoSelectIfProfitable(SDNode *BO, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static SDValue lowerVectorSplatImm(SDNode *Node, unsigned ImmOp, SelectionDAG &DAG, bool IsSigned=false)
const MCPhysReg ArgGPRs[]
static SDValue lowerVECTOR_SHUFFLE_XVPERM(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVPERM (if possible).
static SDValue lowerVECTOR_SHUFFLE_XVILVL(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVILVL (if possible).
static SDValue performFP_TO_INTCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLE_VPERMI(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into VPERMI (if possible).
static SDValue lowerVECTOR_SHUFFLE_XVEXTRINS(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVEXTRINS (if possible).
static SDValue customLegalizeToWOp(SDNode *N, SelectionDAG &DAG, int NumOp, unsigned ExtOpc=ISD::ANY_EXTEND)
static void replaceVecCondBranchResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget, unsigned ResOp)
#define ASRT_LE_GT_CASE(NAME)
static SDValue lowerVECTOR_SHUFFLE_XVPACKEV(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVPACKEV (if possible).
static SDValue matchDeinterleaveBuildVector(SDValue N, unsigned &StartIndex)
static SDValue performBR_CCCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static void computeZeroableShuffleElements(ArrayRef< int > Mask, SDValue V1, SDValue V2, APInt &KnownUndef, APInt &KnownZero)
Compute whether each element of a shuffle is zeroable.
static SDValue combineFP_ROUND(SDValue N, const SDLoc &DL, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static bool combine_CC(SDValue &LHS, SDValue &RHS, SDValue &CC, const SDLoc &DL, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
static SDValue performCONCAT_VECTORSCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue widenShuffleMask(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
static MachineBasicBlock * emitVecCondBranchPseudo(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
static bool canonicalizeShuffleVectorByLane(const SDLoc &DL, MutableArrayRef< int > Mask, MVT VT, SDValue &V1, SDValue &V2, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Shuffle vectors by lane to generate more optimized instructions.
static SDValue lowerVECTOR_SHUFFLE_XVILVH(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVILVH (if possible).
static SDValue lowerVECTOR_SHUFFLE_XVSHUF(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVSHUF (if possible).
static void replaceCMP_XCHG_128Results(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG)
static SDValue lowerVectorPickVE2GR(SDNode *N, SelectionDAG &DAG, unsigned ResOp)
static SDValue performBITREV_WCombine(SDNode *N, SelectionDAG &DAG, TargetLowering::DAGCombinerInfo &DCI, const LoongArchSubtarget &Subtarget)
static SDValue matchHalfOf128BitLanes(SDValue N, bool isLow)
#define IOCSRRD_CASE(NAME, NODE)
static int matchShuffleAsByteRotate(MVT VT, SDValue &V1, SDValue &V2, ArrayRef< int > Mask)
Attempts to match vector shuffle as byte rotation.
static SDValue lowerVECTOR_SHUFFLE_XVPICKEV(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into XVPICKEV (if possible).
static SDValue lowerVECTOR_SHUFFLE_XVREPLVEI(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
Lower VECTOR_SHUFFLE into XVREPLVEI (if possible).
static int matchShuffleAsShift(MVT &ShiftVT, unsigned &Opcode, unsigned ScalarSizeInBits, ArrayRef< int > Mask, int MaskOffset, const APInt &Zeroable)
Attempts to match a shuffle mask against the VBSLL, VBSRL, VSLLI and VSRLI instruction.
static SDValue lowerVECTOR_SHUFFLE_VILVL(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VILVL (if possible).
static SDValue lowerVectorBitClearImm(SDNode *Node, SelectionDAG &DAG)
static MachineBasicBlock * emitBuildPairF64Pseudo(MachineInstr &MI, MachineBasicBlock *BB, const LoongArchSubtarget &Subtarget)
static SDValue lowerVECTOR_SHUFFLEAsLanePermuteAndShuffle(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE as lane permute and then shuffle (if possible).
static void replaceINTRINSIC_WO_CHAINResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG, const LoongArchSubtarget &Subtarget)
#define CSR_CASE(ID)
static SDValue MergeBlocksConvert(SDNode *N, SelectionDAG &DAG, unsigned Opcode, unsigned BlockBits)
static SDValue lowerVECTOR_SHUFFLE_VPICKOD(const SDLoc &DL, ArrayRef< int > Mask, MVT VT, SDValue V1, SDValue V2, SelectionDAG &DAG)
Lower VECTOR_SHUFFLE into VPICKOD (if possible).
static Intrinsic::ID getIntrinsicForMaskedAtomicRMWBinOp(unsigned GRLen, AtomicRMWInst::BinOp BinOp)
static void translateSetCCForBranch(const SDLoc &DL, SDValue &LHS, SDValue &RHS, ISD::CondCode &CC, SelectionDAG &DAG)
static Register allocateArgGPR(CCState &State)
static bool isRepeatedShuffleMask(unsigned LaneSizeInBits, MVT VT, ArrayRef< int > Mask, SmallVectorImpl< int > &RepeatedMask)
Test whether a shuffle mask is equivalent within each sub-lane.
static SDValue convertRMEncoding(SelectionDAG &DAG, const SDLoc &DL, MVT GRLenVT, SDValue RMValue)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
static CodeModel::Model getCodeModel(const PPCSubtarget &S, const TargetMachine &TM, const MachineOperand &MO)
static constexpr MCPhysReg SPReg
const SmallVectorImpl< MachineOperand > & Cond
static bool isValid(const char C)
Returns true if C is a valid mangled character: <0-9a-zA-Z_>.
This file defines the SmallSet class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static bool inRange(const MCExpr *Expr, int64_t MinValue, int64_t MaxValue, bool AllowSymbol=false)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static bool isSequentialOrUndefInRange(ArrayRef< int > Mask, unsigned Pos, unsigned Size, int Low, int Step=1)
Return true if every element in Mask, beginning from position Pos and ending in Pos + Size,...
Value * RHS
Value * LHS
bool isZero() const
Definition APFloat.h:1579
LLVM_READONLY bool isOne() const
Definition APFloat.h:1661
APInt bitcastToAPInt() const
Definition APFloat.h:1475
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:225
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
bool isSignedIntN(unsigned N) const
Check if this APInt has an N-bits signed integer value.
Definition APInt.h:431
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1261
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getZero(unsigned numBits)
Get the '0' value for the specified bit-width.
Definition APInt.h:196
static APInt getBitsSetFrom(unsigned numBits, unsigned loBit)
Constructs an APInt value that has a contiguous range of bits set.
Definition APInt.h:282
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
unsigned getArgNo() const
Return the index of this formal argument in its containing function.
Definition Argument.h:50
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
An instruction that atomically checks whether a specified value is in a memory location,...
AtomicOrdering getFailureOrdering() const
Returns the failure ordering constraint of this cmpxchg instruction.
an instruction that atomically reads a memory location, combines it with another value,...
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ Min
*p = old <signed v ? old : v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ UMax
*p = old >unsigned v ? old : v
@ UDecWrap
Decrement one until a minimum value or zero.
@ Nand
*p = ~(old & v)
Value * getPointerOperand()
bool isFloatingPointOperation() const
BinOp getOperation() const
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
AtomicOrdering getOrdering() const
Returns the ordering constraint of this rmw instruction.
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
bool test(unsigned Idx) const
Returns true if bit Idx is set.
Definition BitVector.h:482
size_type count() const
Returns the number of bits which are set.
Definition BitVector.h:181
A "pseudo-class" with methods for operating on BUILD_VECTORs.
CCState - This class holds information needed while lowering arguments and return values.
unsigned getFirstUnallocated(ArrayRef< MCPhysReg > Regs) const
getFirstUnallocated - Return the index of the first unallocated register in the set,...
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
static CCValAssign getPending(unsigned ValNo, MVT ValVT, MVT LocVT, LocInfo HTP, unsigned ExtraInfo=0)
Register getLocReg() const
LocInfo getLocInfo() const
static CCValAssign getReg(unsigned ValNo, MVT ValVT, MCRegister Reg, MVT LocVT, LocInfo HTP, bool IsCustom=false)
static CCValAssign getCustomReg(unsigned ValNo, MVT ValVT, MCRegister Reg, MVT LocVT, LocInfo HTP)
static CCValAssign getMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP, bool IsCustom=false)
bool needsCustom() const
int64_t getLocMemOffset() const
unsigned getValNo() const
static CCValAssign getCustomMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP)
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
LLVM_ABI bool isMustTailCall() const
Tests if this call site must be tail call optimized.
iterator_range< User::op_iterator > args()
Iteration adapter for range-for loops.
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
const APFloat & getValueAPF() const
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isMinusOne() const
This function will return true iff every bit in this constant is set to true.
Definition Constants.h:231
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
Definition Constants.h:219
uint64_t getZExtValue() const
int64_t getSExtValue() const
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
unsigned getPointerSizeInBits(unsigned AS=0) const
The size in bits of the pointer representation in a given address space.
Definition DataLayout.h:501
LLVM_ABI Align getPrefTypeAlign(Type *Ty) const
Returns the preferred stack/global alignment for the specified type.
A debug info location.
Definition DebugLoc.h:126
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
iterator_range< arg_iterator > args()
Definition Function.h:877
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:769
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:781
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
Argument * getArg(unsigned i) const
Definition Function.h:871
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
bool isDSOLocal() const
Common base class shared among various IRBuilders.
Definition IRBuilder.h:111
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2918
LLVM_ABI const Module * getModule() const
Return the module owning the function this instruction belongs to or nullptr it the function does not...
LLVM_ABI InstListType::iterator eraseFromParent()
This method unlinks 'this' from the containing basic block and deletes it.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
Class to represent integer types.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
ISD::LoadExtType getExtensionType() const
Return whether this is a plain node, or one of the varieties of value-extending loads.
LoongArchMachineFunctionInfo - This class is derived from MachineFunctionInfo and contains private Lo...
void setIncomingIndirectArg(unsigned ArgIndex, Register Reg)
Register getIncomingIndirectArg(unsigned ArgIndex) const
const LoongArchRegisterInfo * getRegisterInfo() const override
const LoongArchTargetLowering * getTargetLowering() const override
const LoongArchInstrInfo * getInstrInfo() const override
bool isUsedByReturnOnly(SDNode *N, SDValue &Chain) const override
Return true if result of the specified node is used by a return node only.
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
Register getExceptionPointerRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception address on entry to an ...
SDValue getSqrtEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps, bool &UseOneConstNR, bool Reciprocal) const override
Hooks for building estimates in place of slower divisions and square roots.
bool isLegalICmpImmediate(int64_t Imm) const override
Return true if the specified immediate is legal icmp immediate, that is the target has icmp instructi...
TargetLowering::AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *CI) const override
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
Value * emitMaskedAtomicCmpXchgIntrinsic(IRBuilderBase &Builder, AtomicCmpXchgInst *CI, Value *AlignedAddr, Value *CmpVal, Value *NewVal, Value *Mask, AtomicOrdering Ord) const override
Perform a masked cmpxchg using a target-specific intrinsic.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
Return the ValueType of the result of SETCC operations.
std::pair< bool, uint64_t > isImmVLDILegalForMode1(const APInt &SplatValue, const unsigned SplatBitSize) const
Check if a constant splat can be generated using [x]vldi, where imm[12] is 1.
void getTgtMemIntrinsic(SmallVectorImpl< IntrinsicInfo > &Infos, const CallBase &I, MachineFunction &MF, unsigned Intrinsic) const override
Given an intrinsic, checks if on the target the intrinsic will need to map to a MemIntrinsicNode (tou...
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
bool hasInlineStackProbe(const MachineFunction &MF) const override
True if stack clash protection is enabled for this function.
SDValue LowerCall(TargetLowering::CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
Register getExceptionSelectorRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception typeid on entry to a la...
bool decomposeMulByConstant(LLVMContext &Context, EVT VT, SDValue C) const override
Return true if it is profitable to transform an integer multiplication-by-constant into simpler opera...
bool isExtractVecEltCheap(EVT VT, unsigned Index) const override
Return true if extraction of a scalar element from the given vector type at the given index is cheap.
LegalizeTypeAction getPreferredVectorAction(MVT VT) const override
Return the preferred vector type legalization action.
bool isSExtCheaperThanZExt(EVT SrcVT, EVT DstVT) const override
Return true if sign-extension from FromTy to ToTy is cheaper than zero-extension.
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const override
Determine if the target supports unaligned memory accesses.
bool isCheapToSpeculateCtlz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic ctlz.
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
bool shouldAlignPointerArgs(CallInst *CI, unsigned &MinSize, Align &PrefAlign) const override
Return true if the pointer arguments to CI should be aligned by aligning the object whose address is ...
Value * emitMaskedAtomicRMWIntrinsic(IRBuilderBase &Builder, AtomicRMWInst *AI, Value *AlignedAddr, Value *Incr, Value *Mask, Value *ShiftAmt, AtomicOrdering Ord) const override
Perform a masked atomicrmw using a target-specific intrinsic.
bool isZExtFree(SDValue Val, EVT VT2) const override
Return true if zero-extending the specific node Val to type VT2 is free (either because it's implicit...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool signExtendConstant(const ConstantInt *CI) const override
Return true if this constant should be sign extended when promoting to a larger type.
ExtractSubvectorCost getExtractSubvectorCost(EVT ResVT, EVT SrcVT, unsigned Index) const override
Return the cost of extracting a subvector of type ResVT from a vector of type SrcVT,...
TargetLowering::AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *AI) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
bool isLegalAddImmediate(int64_t Imm) const override
Return true if the specified immediate is legal add immediate, that is the target has add instruction...
bool isCheapToSpeculateCttz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic cttz.
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
MachineBasicBlock * emitDynamicProbedAlloc(MachineInstr &MI, MachineBasicBlock *MBB) const
bool shouldSignExtendTypeInLibCall(Type *Ty, bool IsSigned) const override
Returns true if arguments should be sign-extended in lib calls.
bool shouldScalarizeBinop(SDValue VecOp) const override
Try to convert an extract element of a vector binary operation into an extract element followed by a ...
bool isFPImmVLDILegal(const APFloat &Imm, EVT VT) const
bool shouldExtendTypeInLibCall(EVT Type) const override
Returns true if arguments should be extended in lib calls.
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
bool hasAndNot(SDValue Y) const override
Return true if the target has a bitwise and-not operation: X = ~A & B This can be used to simplify se...
bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override
Return true if folding a constant offset with the given GlobalAddress is legal.
unsigned getStackProbeSize(const MachineFunction &MF, Align StackAlign) const
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth) const override
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
void emitExpandAtomicRMW(AtomicRMWInst *AI) const override
Perform a atomicrmw expansion using a target-specific way.
ISD::NodeType getExtendForAtomicCmpSwapArg() const override
Returns how the platform's atomic compare and swap expects its comparison value to be extended (ZERO_...
LoongArchTargetLowering(const TargetMachine &TM, const LoongArchSubtarget &STI)
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
bool hasAndNotCompare(SDValue Y) const override
Return true if the target should transform: (X & Y) == Y ---> (~X & Y) == 0 (X & Y) !...
SDValue getRecipEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps) const override
Return a reciprocal estimate value for the input operand.
bool canMergeStoresTo(unsigned AddressSpace, EVT MemVT, const MachineFunction &MF) const override
Returns if it's reasonable to merge stores to MemVT size.
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
bool mayBeEmittedAsTailCall(const CallInst *CI) const override
Return true if the target may be able emit the call instruction as a tail call.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
bool hasFeature(unsigned Feature) const
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
bool is128BitVector() const
Return true if this is a 128-bit vector type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
bool is256BitVector() const
Return true if this is a 256-bit vector type.
bool isScalarInteger() const
Return true if this is an integer, not including vectors.
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
static MVT getIntegerVT(unsigned BitWidth)
MVT getDoubleNumVectorElementsVT() const
MVT getHalfNumVectorElementsVT() const
Return a VT for a vector type with the same element type but half the number of elements.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
MVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
void push_back(MachineInstr *MI)
void setCallFrameSize(unsigned N)
Set the call frame size on entry to this basic block.
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
void setFrameAddressIsTaken(bool T)
void setHasTailCall(bool V=true)
void setReturnAddressIsTaken(bool s)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
bool isImplicitDef() const
LLVM_ABI void collectDebugValues(SmallVectorImpl< MachineInstr * > &DbgValues)
Scan instructions immediately following MI and collect any matching DBG_VALUEs.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI MachineInstrBundleIterator< MachineInstr > eraseFromParent()
Unlink 'this' from the containing basic block and delete it.
A description of a memory reference used in the backend.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
Flags getFlags() const
Return the raw flags of the source value,.
MachineOperand class - Representation of each machine instruction operand.
void setIsKill(bool Val=true)
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
Align getAlign() const
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const MachinePointerInfo & getPointerInfo() const
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
Represent a mutable reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:294
Class to represent pointers.
unsigned getAddressSpace() const
Return the address space of the Pointer type.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
const APInt & getAsAPIntVal() const
Helper method returns the APInt value of a ConstantSDNode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
LLVM_ABI bool isOnlyUserOf(const SDNode *N) const
Return true if this node is the only use of N.
size_t use_size() const
Return the number of uses of this node.
MVT getSimpleValueType(unsigned ResNo) const
Return the type of a specified result as a simple type.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumOperands() const
Return the number of values used by this operation.
const SDValue & getOperand(unsigned Num) const
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
bool isUndef() const
Returns true if the node type is UNDEF or POISON.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
uint64_t getScalarValueSizeInBits() const
uint64_t getConstantOperandVal(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue FoldSetCC(EVT VT, SDValue N1, SDValue N2, ISD::CondCode Cond, const SDLoc &dl, SDNodeFlags Flags={})
Constant fold a setcc to true or false.
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
bool isSafeToSpeculativelyExecute(unsigned Opcode) const
Some opcodes may create immediate undefined behavior when used with some values (integer division-by-...
SDValue getExtractSubvector(const SDLoc &DL, EVT VT, SDValue Vec, unsigned Idx)
Return the VT typed sub-vector of Vec at Idx.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getInsertSubvector(const SDLoc &DL, SDValue Vec, SDValue SubVec, unsigned Idx)
Insert SubVec at the Idx element of Vec.
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
static constexpr unsigned MaxRecursionDepth
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI bool isSplatValue(SDValue V, const APInt &DemandedElts, APInt &UndefElts, unsigned Depth=0) const
Test whether V has a splatted value for all the demanded elements.
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI SDValue getNegative(SDValue Val, const SDLoc &DL, EVT VT)
Create negative operation as (SUB 0, Val).
LLVM_ABI void setNodeMemRefs(MachineSDNode *N, ArrayRef< MachineMemOperand * > NewMemRefs)
Mutate the specified machine node's memory references to the provided list.
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getCommutedVectorShuffle(const ShuffleVectorSDNode &SV)
Returns an ISD::VECTOR_SHUFFLE node semantically equivalent to the shuffle node in input but with swa...
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI std::pair< SDValue, SDValue > SplitVector(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the vector with EXTRACT_SUBVECTOR using the provided VTs and return the low/high part.
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI SDValue FoldConstantArithmetic(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDValue > Ops, SDNodeFlags Flags=SDNodeFlags())
LLVM_ABI SDValue getExternalSymbol(const char *Sym, EVT VT)
const TargetMachine & getTarget() const
LLVM_ABI SDValue WidenVector(const SDValue &N, const SDLoc &DL)
Widen the vector up to the next power of two using INSERT_SUBVECTOR.
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI SDValue getCondCode(ISD::CondCode Cond)
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue CreateStackTemporary(TypeSize Bytes, Align Alignment)
Create a stack temporary based on the size in bytes and the alignment.
SDValue getTargetConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offset=0, unsigned TargetFlags=0)
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
SDValue getSplat(EVT VT, const SDLoc &DL, SDValue Op)
Returns a node representing a splat of one value into all lanes of the provided vector type.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
static LLVM_ABI bool isReverseMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask swaps the order of elements from exactly one source vector.
ArrayRef< int > getMask() const
SmallSet - This maintains a set of unique values, optimizing for the case when the set is small (less...
Definition SmallSet.h:134
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
Definition SmallSet.h:176
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void assign(size_type NumElts, ValueParamT Elt)
void reserve(size_type N)
typename SuperClass::const_iterator const_iterator
iterator insert(iterator I, T &&Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
StackOffset holds a fixed and a scalable offset in bytes.
Definition TypeSize.h:30
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
TargetInstrInfo - Interface to description of machine instruction set.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
MachineBasicBlock * emitPatchPoint(MachineInstr &MI, MachineBasicBlock *MBB) const
Replace/modify any TargetFrameIndex operands with a targte-dependent sequence of memory operands that...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual bool isZExtFree(Type *FromTy, Type *ToTy) const
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
void setMaxBytesForAlignment(unsigned MaxBytes)
bool isOperationLegalOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal using promotion.
void setPrefLoopAlignment(Align Alignment)
Set the target's preferred loop alignment.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
virtual TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const
Return the preferred vector type legalization action.
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
virtual EVT getTypeToTransformTo(LLVMContext &Context, EVT VT) const
For types supported by the target, this is an identity function.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
ExtractSubvectorCost
Enum that specifies how expensive lowering an EXTRACT_SUBVECTOR is.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
virtual bool isBinOp(unsigned Opcode) const
Return true if the node is a math/logic binary operator.
void setMinCmpXchgSizeInBits(unsigned SizeInBits)
Sets the minimum cmpxchg or ll/sc size supported by the backend.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setCondCodeAction(ArrayRef< ISD::CondCode > CCs, MVT VT, LegalizeAction Action)
Indicate that the specified condition code is or isn't supported on the target and indicate what to d...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
LegalizeTypeAction getTypeAction(LLVMContext &Context, EVT VT) const
Return how we should legalize values of this type, either it is already legal (return 'Legal') or we ...
std::vector< ArgListEntry > ArgListTy
bool isOperationLegalOrCustomOrPromote(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
bool SimplifyDemandedVectorElts(SDValue Op, const APInt &DemandedEltMask, APInt &KnownUndef, APInt &KnownZero, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Vector Op.
virtual InlineAsm::ConstraintCode getInlineAsmMemConstraint(StringRef ConstraintCode) const
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
virtual bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0) const
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
TargetLowering(const TargetLowering &)=delete
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
Primary interface to the complete machine description for the target machine.
bool useTLSDESC() const
Returns true if this target uses TLS Descriptors.
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
bool shouldAssumeDSOLocal(const GlobalValue *GV) const
CodeModel::Model getCodeModel() const
Returns the code model.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetInstrInfo * getInstrInfo() const
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getIntegerBitWidth() const
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
static LLVM_ABI IntegerType * getIntNTy(LLVMContext &C, unsigned N)
Definition Type.cpp:303
This class is used to represent EVT's, which are used to parameterize some operations.
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
self_iterator getIterator()
Definition ilist_node.h:123
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ PreserveMost
Used for runtime calls that preserves most registers.
Definition CallingConv.h:63
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ PreserveNone
Used for runtime calls that preserves none general registers.
Definition CallingConv.h:90
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
LLVM_ABI bool isConstantSplatVectorAllOnes(const SDNode *N, bool BuildVectorOnly=false)
Return true if the specified node is a BUILD_VECTOR or SPLAT_VECTOR where all of the elements are ~0 ...
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ STRICT_FSETCC
STRICT_FSETCC/STRICT_FSETCCS - Constrained versions of SETCC, used for floating-point operands only.
Definition ISDOpcodes.h:516
@ DELETED_NODE
DELETED_NODE - This is an illegal value that is used to catch errors.
Definition ISDOpcodes.h:47
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ GlobalAddress
Definition ISDOpcodes.h:90
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:589
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ MEMBARRIER
MEMBARRIER - Compiler barrier only; generate a no-op.
@ ATOMIC_FENCE
OUTCHAIN = ATOMIC_FENCE(INCHAIN, ordering, scope) This corresponds to the fence instruction.
@ SIGN_EXTEND_VECTOR_INREG
SIGN_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register sign-extension of the low ...
Definition ISDOpcodes.h:928
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:91
@ SET_ROUNDING
Set rounding mode.
Definition ISDOpcodes.h:993
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ AVGCEILS
AVGCEILS/AVGCEILU - Rounding averaging add - Add two integers using an integer of type i[N+2],...
Definition ISDOpcodes.h:725
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:546
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:553
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ VECREDUCE_ADD
Integer reductions may have a result type larger than the vector element type.
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
Definition ISDOpcodes.h:988
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ READ_REGISTER
READ_REGISTER, WRITE_REGISTER - This node represents llvm.register on the DAG, which implements the n...
Definition ISDOpcodes.h:141
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ EH_DWARF_CFA
EH_DWARF_CFA - This node represents the pointer to the DWARF Canonical Frame Address (CFA),...
Definition ISDOpcodes.h:152
@ BF16_TO_FP
BF16_TO_FP, FP_TO_BF16 - These operators are used to perform promotions and truncation for bfloat16.
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:112
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ AVGFLOORS
AVGFLOORS/AVGFLOORU - Averaging add - Add two integers using an integer of type i[N+1],...
Definition ISDOpcodes.h:720
@ FREEZE
FREEZE - FREEZE(VAL) returns an arbitrary value if VAL is UNDEF (or is evaluated to UNDEF),...
Definition ISDOpcodes.h:243
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:939
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ VAARG
VAARG - VAARG has four operands: an input chain, a pointer, a SRCVALUE, and the alignment.
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
@ ABDS
ABDS/ABDU - Absolute difference - Return the absolute difference between two numbers interpreted as s...
Definition ISDOpcodes.h:732
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
bool isExtVecInRegOpcode(unsigned Opcode)
LLVM_ABI bool isConstantSplatVectorAllZeros(const SDNode *N, bool BuildVectorOnly=false)
Return true if the specified node is a BUILD_VECTOR or SPLAT_VECTOR where all of the elements are 0 o...
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
bool isBitwiseLogicOp(unsigned Opcode)
Whether this is bitwise logic opcode.
LLVM_ABI bool isFreezeUndef(const SDNode *N)
Return true if the specified node is FREEZE(UNDEF).
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LLVM_ABI bool isBuildVectorAllOnes(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are ~0 or undef.
LLVM_ABI NodeType getVecReduceBaseOpcode(unsigned VecReduceOpcode)
Get underlying scalar opcode for VECREDUCE opcode.
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
bool isIntEqualitySetCC(CondCode Code)
Return true if this is a setcc instruction that performs an equality comparison when used with intege...
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
ABI getTargetABI(StringRef ABIName)
InstSeq generateInstSeq(int64_t Val)
LLVM_ABI Libcall getSINTTOFP(EVT OpVT, EVT RetVT)
getSINTTOFP - Return the SINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getUINTTOFP(EVT OpVT, EVT RetVT)
getUINTTOFP - Return the UINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPTOSINT(EVT OpVT, EVT RetVT)
getFPTOSINT - Return the FPTOSINT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPROUND(EVT OpVT, EVT RetVT)
getFPROUND - Return the FPROUND_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
ValuesClass values(OptsTy... Options)
Helper to build a ValuesClass by forwarding a variable number of arguments as an initializer list to ...
initializer< Ty > init(const Ty &Val)
Sequence
A sequence of states that a pointer may go through in which an objc_retain and objc_release are actua...
Definition PtrState.h:41
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
@ Undef
Value of the register doesn't matter.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
constexpr RegState getKillRegState(bool B)
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool isIntOrFPConstant(SDValue V)
Return true if V is either a integer or FP constant.
int bit_width(T Value)
Returns the number of bits needed to represent Value if Value is nonzero.
Definition bit.h:325
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool widenShuffleMaskElts(int Scale, ArrayRef< int > Mask, SmallVectorImpl< int > &ScaledMask)
Try to transform a shuffle mask by replacing elements with the scaled index for an equivalent mask of...
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr bool isMask_64(uint64_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:262
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Other
Any other memory.
Definition ModRef.h:68
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr bool isShiftedInt(int64_t x)
Checks if a signed integer is an N bit number shifted left by S.
Definition MathExtras.h:183
constexpr unsigned BitWidth
std::string join_items(Sep Separator, Args &&... Items)
Joins the strings in the parameter pack Items, adding Separator between the elements....
ExceptionHandling
Definition CodeGen.h:54
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
PointerUnion< const Value *, const PseudoSourceValue * > ValueType
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
static EVT getFloatingPointVT(unsigned BitWidth)
Returns the EVT that represents a floating-point type with the given number of bits.
Definition ValueTypes.h:55
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool is256BitVector() const
Return true if this is a 256-bit vector type.
Definition ValueTypes.h:235
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
EVT getHalfNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:484
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
Align getNonZeroOrigAlign() const
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
static LLVM_ABI MachinePointerInfo getUnknownStack(MachineFunction &MF)
Stack memory without other information.
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
SmallVector< ISD::OutputArg, 32 > Outs
LLVM_ABI SDValue CombineTo(SDNode *N, ArrayRef< SDValue > To, bool AddTo=true)
This structure is used to pass arguments to makeLibCall function.
MakeLibCallOptions & setTypeListBeforeSoften(ArrayRef< EVT > OpsVT, EVT RetVT)
A convenience struct that encapsulates a DAG, and two SDValues for returning information from TargetL...