LLVM 24.0.0git
ARMISelLowering.cpp
Go to the documentation of this file.
1//===- ARMISelLowering.cpp - ARM DAG Lowering Implementation --------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines the interfaces that ARM uses to lower LLVM code into a
10// selection DAG.
11//
12//===----------------------------------------------------------------------===//
13
14#include "ARMISelLowering.h"
15#include "ARMBaseInstrInfo.h"
16#include "ARMBaseRegisterInfo.h"
17#include "ARMCallingConv.h"
20#include "ARMPerfectShuffle.h"
21#include "ARMRegisterInfo.h"
22#include "ARMSelectionDAGInfo.h"
23#include "ARMSubtarget.h"
27#include "Utils/ARMBaseInfo.h"
28#include "llvm/ADT/APFloat.h"
29#include "llvm/ADT/APInt.h"
30#include "llvm/ADT/ArrayRef.h"
31#include "llvm/ADT/BitVector.h"
32#include "llvm/ADT/DenseMap.h"
33#include "llvm/ADT/STLExtras.h"
36#include "llvm/ADT/Statistic.h"
38#include "llvm/ADT/StringRef.h"
40#include "llvm/ADT/Twine.h"
66#include "llvm/IR/Attributes.h"
67#include "llvm/IR/CallingConv.h"
68#include "llvm/IR/Constant.h"
69#include "llvm/IR/Constants.h"
70#include "llvm/IR/DataLayout.h"
71#include "llvm/IR/DebugLoc.h"
73#include "llvm/IR/Function.h"
74#include "llvm/IR/GlobalAlias.h"
75#include "llvm/IR/GlobalValue.h"
77#include "llvm/IR/IRBuilder.h"
78#include "llvm/IR/InlineAsm.h"
79#include "llvm/IR/Instruction.h"
82#include "llvm/IR/Intrinsics.h"
83#include "llvm/IR/IntrinsicsARM.h"
84#include "llvm/IR/Module.h"
85#include "llvm/IR/Type.h"
86#include "llvm/IR/User.h"
87#include "llvm/IR/Value.h"
88#include "llvm/MC/MCInstrDesc.h"
90#include "llvm/MC/MCSchedule.h"
97#include "llvm/Support/Debug.h"
105#include <algorithm>
106#include <cassert>
107#include <cstdint>
108#include <iterator>
109#include <limits>
110#include <optional>
111#include <tuple>
112#include <utility>
113#include <vector>
114
115using namespace llvm;
116
117#define DEBUG_TYPE "arm-isel"
118
119STATISTIC(NumTailCalls, "Number of tail calls");
120STATISTIC(NumOptimizedImms, "Number of times immediates were optimized");
121STATISTIC(NumMovwMovt, "Number of GAs materialized with movw + movt");
122STATISTIC(NumLoopByVals, "Number of loops generated for byval arguments");
123STATISTIC(NumConstpoolPromoted,
124 "Number of constants with their storage promoted into constant pools");
125
126static cl::opt<bool>
127ARMInterworking("arm-interworking", cl::Hidden,
128 cl::desc("Enable / disable ARM interworking (for debugging only)"),
129 cl::init(true));
130
132 "arm-promote-constant", cl::Hidden,
133 cl::desc("Enable / disable promotion of unnamed_addr constants into "
134 "constant pools"),
135 cl::init(false)); // FIXME: set to true by default once PR32780 is fixed
137 "arm-promote-constant-max-size", cl::Hidden,
138 cl::desc("Maximum size of constant to promote into a constant pool"),
139 cl::init(64));
141 "arm-promote-constant-max-total", cl::Hidden,
142 cl::desc("Maximum size of ALL constants to promote into a constant pool"),
143 cl::init(128));
144
146MVEMaxSupportedInterleaveFactor("mve-max-interleave-factor", cl::Hidden,
147 cl::desc("Maximum interleave factor for MVE VLDn to generate."),
148 cl::init(2));
149
151 "arm-max-base-updates-to-check", cl::Hidden,
152 cl::desc("Maximum number of base-updates to check generating postindex."),
153 cl::init(64));
154
155/// Value type used for "flags" operands / results (either CPSR or FPSCR_NZCV).
156constexpr MVT FlagsVT = MVT::i32;
157
158// The APCS parameter registers.
159static const MCPhysReg GPRArgRegs[] = {
160 ARM::R0, ARM::R1, ARM::R2, ARM::R3
161};
162
164 SelectionDAG &DAG, const SDLoc &DL) {
166 assert(Arg.ArgVT.bitsLT(MVT::i32));
167 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, Arg.ArgVT, Value);
168 SDValue Ext =
170 MVT::i32, Trunc);
171 return Ext;
172}
173
174void ARMTargetLowering::addTypeForNEON(MVT VT, MVT PromotedLdStVT) {
175 if (VT != PromotedLdStVT) {
177 AddPromotedToType (ISD::LOAD, VT, PromotedLdStVT);
178
180 AddPromotedToType (ISD::STORE, VT, PromotedLdStVT);
181 }
182
183 MVT ElemTy = VT.getVectorElementType();
184 if (ElemTy != MVT::f64)
188 if (ElemTy == MVT::i32) {
193 } else {
198 }
207 if (VT.isInteger()) {
211 }
212
213 // Neon does not support vector divide/remainder operations.
222
223 if (!VT.isFloatingPoint() && VT != MVT::v2i64 && VT != MVT::v1i64)
224 for (auto Opcode : {ISD::ABS, ISD::ABDS, ISD::ABDU, ISD::SMIN, ISD::SMAX,
226 setOperationAction(Opcode, VT, Legal);
227 if (!VT.isFloatingPoint())
228 for (auto Opcode : {ISD::SADDSAT, ISD::UADDSAT, ISD::SSUBSAT, ISD::USUBSAT})
229 setOperationAction(Opcode, VT, Legal);
230}
231
232void ARMTargetLowering::addDRTypeForNEON(MVT VT) {
233 addRegisterClass(VT, &ARM::DPRRegClass);
234 addTypeForNEON(VT, MVT::f64);
235}
236
237void ARMTargetLowering::addQRTypeForNEON(MVT VT) {
238 addRegisterClass(VT, &ARM::DPairRegClass);
239 addTypeForNEON(VT, MVT::v2f64);
240}
241
242void ARMTargetLowering::setAllExpand(MVT VT) {
243 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
245
246 // We support these really simple operations even on types where all
247 // the actual arithmetic has to be broken down into simpler
248 // operations or turned into library calls.
253}
254
255void ARMTargetLowering::addAllExtLoads(const MVT From, const MVT To,
256 LegalizeAction Action) {
257 setLoadExtAction(ISD::EXTLOAD, From, To, Action);
258 setLoadExtAction(ISD::ZEXTLOAD, From, To, Action);
259 setLoadExtAction(ISD::SEXTLOAD, From, To, Action);
260}
261
262void ARMTargetLowering::addMVEVectorTypes(bool HasMVEFP) {
263 const MVT IntTypes[] = { MVT::v16i8, MVT::v8i16, MVT::v4i32 };
264
265 for (auto VT : IntTypes) {
266 addRegisterClass(VT, &ARM::MQPRRegClass);
297
298 // No native support for these.
308
309 // Vector reductions
319
320 if (!HasMVEFP) {
325 } else {
328 }
329
330 // Pre and Post inc are supported on loads and stores
331 for (unsigned im = (unsigned)ISD::PRE_INC;
332 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) {
337 }
338 }
339
340 const MVT FloatTypes[] = { MVT::v8f16, MVT::v4f32 };
341 for (auto VT : FloatTypes) {
342 addRegisterClass(VT, &ARM::MQPRRegClass);
343 if (!HasMVEFP)
344 setAllExpand(VT);
345
346 // These are legal or custom whether we have MVE.fp or not
359
360 // Pre and Post inc are supported on loads and stores
361 for (unsigned im = (unsigned)ISD::PRE_INC;
362 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) {
367 }
368
369 if (HasMVEFP) {
377 }
382
383 // No native support for these.
398 }
399 }
400
401 // Custom Expand smaller than legal vector reductions to prevent false zero
402 // items being added.
411
412 // We 'support' these types up to bitcast/load/store level, regardless of
413 // MVE integer-only / float support. Only doing FP data processing on the FP
414 // vector types is inhibited at integer-only level.
415 const MVT LongTypes[] = { MVT::v2i64, MVT::v2f64 };
416 for (auto VT : LongTypes) {
417 addRegisterClass(VT, &ARM::MQPRRegClass);
418 setAllExpand(VT);
424 }
426
427 // We can do bitwise operations on v2i64 vectors
428 setOperationAction(ISD::AND, MVT::v2i64, Legal);
429 setOperationAction(ISD::OR, MVT::v2i64, Legal);
430 setOperationAction(ISD::XOR, MVT::v2i64, Legal);
431
432 // It is legal to extload from v4i8 to v4i16 or v4i32.
433 addAllExtLoads(MVT::v8i16, MVT::v8i8, Legal);
434 addAllExtLoads(MVT::v4i32, MVT::v4i16, Legal);
435 addAllExtLoads(MVT::v4i32, MVT::v4i8, Legal);
436
437 // It is legal to sign extend from v4i8/v4i16 to v4i32 or v8i8 to v8i16.
443
444 // Some truncating stores are legal too.
445 setTruncStoreAction(MVT::v4i32, MVT::v4i16, Legal);
446 setTruncStoreAction(MVT::v4i32, MVT::v4i8, Legal);
447 setTruncStoreAction(MVT::v8i16, MVT::v8i8, Legal);
448
449 // Pre and Post inc on these are legal, given the correct extends
450 for (unsigned im = (unsigned)ISD::PRE_INC;
451 im != (unsigned)ISD::LAST_INDEXED_MODE; ++im) {
452 for (auto VT : {MVT::v8i8, MVT::v4i8, MVT::v4i16}) {
457 }
458 }
459
460 // Predicate types
461 const MVT pTypes[] = {MVT::v16i1, MVT::v8i1, MVT::v4i1, MVT::v2i1};
462 for (auto VT : pTypes) {
463 addRegisterClass(VT, &ARM::VCCRRegClass);
478
479 if (!HasMVEFP) {
484 }
485 }
489 setOperationAction(ISD::OR, MVT::v2i1, Expand);
495
504}
505
507 return static_cast<const ARMBaseTargetMachine &>(getTargetMachine());
508}
509
511 const ARMSubtarget &STI)
512 : TargetLowering(TM_, STI), Subtarget(&STI),
513 RegInfo(Subtarget->getRegisterInfo()),
514 Itins(Subtarget->getInstrItineraryData()) {
515 const auto &TM = static_cast<const ARMBaseTargetMachine &>(TM_);
516
519
520 const Triple &TT = TM.getTargetTriple();
521
522 if (Subtarget->isThumb1Only())
523 addRegisterClass(MVT::i32, &ARM::tGPRRegClass);
524 else
525 addRegisterClass(MVT::i32, &ARM::GPRRegClass);
526
527 if (!Subtarget->useSoftFloat() && !Subtarget->isThumb1Only() &&
528 Subtarget->hasFPRegs()) {
529 addRegisterClass(MVT::f32, &ARM::SPRRegClass);
530 addRegisterClass(MVT::f64, &ARM::DPRRegClass);
531
532 if (!Subtarget->hasVFP2Base()) {
533 setAllExpand(MVT::f32);
534 } else {
537
540 setOperationAction(Op, MVT::f32, Legal);
541 }
542 if (!Subtarget->hasFP64()) {
543 setAllExpand(MVT::f64);
544 } else {
547 setOperationAction(Op, MVT::f64, Legal);
548
550 }
551 }
552
553 if (Subtarget->hasFullFP16()) {
556 setOperationAction(Op, MVT::f16, Legal);
557
558 addRegisterClass(MVT::f16, &ARM::HPRRegClass);
561
566 }
567
568 if (Subtarget->hasBF16()) {
569 addRegisterClass(MVT::bf16, &ARM::HPRRegClass);
570 setAllExpand(MVT::bf16);
571 if (!Subtarget->hasFullFP16())
575 } else {
580 }
581
583 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) {
584 setTruncStoreAction(VT, InnerVT, Expand);
585 addAllExtLoads(VT, InnerVT, Expand);
586 }
587
590
592 }
593
594 if (!Subtarget->isThumb1Only() && !Subtarget->hasV8_1MMainlineOps())
596
597 if (!Subtarget->hasV8_1MMainlineOps())
599
600 if (!Subtarget->isThumb1Only())
602
605
608
609 if (Subtarget->hasMVEIntegerOps())
610 addMVEVectorTypes(Subtarget->hasMVEFloatOps());
611
612 // Combine low-overhead loop intrinsics so that we can lower i1 types.
613 if (Subtarget->hasLOB()) {
615 }
616
617 if (Subtarget->hasNEON()) {
618 addDRTypeForNEON(MVT::v2f32);
619 addDRTypeForNEON(MVT::v8i8);
620 addDRTypeForNEON(MVT::v4i16);
621 addDRTypeForNEON(MVT::v2i32);
622 addDRTypeForNEON(MVT::v1i64);
623
624 addQRTypeForNEON(MVT::v4f32);
625 addQRTypeForNEON(MVT::v2f64);
626 addQRTypeForNEON(MVT::v16i8);
627 addQRTypeForNEON(MVT::v8i16);
628 addQRTypeForNEON(MVT::v4i32);
629 addQRTypeForNEON(MVT::v2i64);
630
631 if (Subtarget->hasFullFP16()) {
632 addQRTypeForNEON(MVT::v8f16);
633 addDRTypeForNEON(MVT::v4f16);
634 }
635
636 if (Subtarget->hasBF16()) {
637 addQRTypeForNEON(MVT::v8bf16);
638 addDRTypeForNEON(MVT::v4bf16);
639 }
640 }
641
642 if (Subtarget->hasMVEIntegerOps() || Subtarget->hasNEON()) {
643 // v2f64 is legal so that QR subregs can be extracted as f64 elements, but
644 // none of Neon, MVE or VFP supports any arithmetic operations on it.
645 setOperationAction(ISD::FADD, MVT::v2f64, Expand);
646 setOperationAction(ISD::FSUB, MVT::v2f64, Expand);
647 setOperationAction(ISD::FMUL, MVT::v2f64, Expand);
648 // FIXME: Code duplication: FDIV and FREM are expanded always, see
649 // ARMTargetLowering::addTypeForNEON method for details.
650 setOperationAction(ISD::FDIV, MVT::v2f64, Expand);
651 setOperationAction(ISD::FREM, MVT::v2f64, Expand);
652 // FIXME: Create unittest.
653 // In another words, find a way when "copysign" appears in DAG with vector
654 // operands.
656 // FIXME: Code duplication: SETCC has custom operation action, see
657 // ARMTargetLowering::addTypeForNEON method for details.
659 // FIXME: Create unittest for FNEG and for FABS.
660 setOperationAction(ISD::FNEG, MVT::v2f64, Expand);
661 setOperationAction(ISD::FABS, MVT::v2f64, Expand);
663 setOperationAction(ISD::FSIN, MVT::v2f64, Expand);
664 setOperationAction(ISD::FCOS, MVT::v2f64, Expand);
665 setOperationAction(ISD::FTAN, MVT::v2f64, Expand);
666 setOperationAction(ISD::FPOW, MVT::v2f64, Expand);
667 setOperationAction(ISD::FLOG, MVT::v2f64, Expand);
670 setOperationAction(ISD::FEXP, MVT::v2f64, Expand);
679 setOperationAction(ISD::FMA, MVT::v2f64, Expand);
680 }
681
682 if (Subtarget->hasNEON()) {
683 // The same with v4f32. But keep in mind that vadd, vsub, vmul are natively
684 // supported for v4f32.
686 setOperationAction(ISD::FSIN, MVT::v4f32, Expand);
687 setOperationAction(ISD::FCOS, MVT::v4f32, Expand);
688 setOperationAction(ISD::FTAN, MVT::v4f32, Expand);
689 setOperationAction(ISD::FPOW, MVT::v4f32, Expand);
690 setOperationAction(ISD::FLOG, MVT::v4f32, Expand);
693 setOperationAction(ISD::FEXP, MVT::v4f32, Expand);
702
703 // Mark v2f32 intrinsics.
705 setOperationAction(ISD::FSIN, MVT::v2f32, Expand);
706 setOperationAction(ISD::FCOS, MVT::v2f32, Expand);
707 setOperationAction(ISD::FTAN, MVT::v2f32, Expand);
708 setOperationAction(ISD::FPOW, MVT::v2f32, Expand);
709 setOperationAction(ISD::FLOG, MVT::v2f32, Expand);
712 setOperationAction(ISD::FEXP, MVT::v2f32, Expand);
721
724 setOperationAction(Op, MVT::v4f16, Expand);
725 setOperationAction(Op, MVT::v8f16, Expand);
726 }
727
728 // Neon does not support some operations on v1i64 and v2i64 types.
729 setOperationAction(ISD::MUL, MVT::v1i64, Expand);
730 // Custom handling for some quad-vector types to detect VMULL.
731 setOperationAction(ISD::MUL, MVT::v8i16, Custom);
732 setOperationAction(ISD::MUL, MVT::v4i32, Custom);
733 setOperationAction(ISD::MUL, MVT::v2i64, Custom);
734 // Custom handling for some vector types to avoid expensive expansions
735 setOperationAction(ISD::SDIV, MVT::v4i16, Custom);
737 setOperationAction(ISD::UDIV, MVT::v4i16, Custom);
739 // Neon does not have single instruction SINT_TO_FP and UINT_TO_FP with
740 // a destination type that is wider than the source, and nor does
741 // it have a FP_TO_[SU]INT instruction with a narrower destination than
742 // source.
751
754
755 // NEON does not have single instruction CTPOP for vectors with element
756 // types wider than 8-bits. However, custom lowering can leverage the
757 // v8i8/v16i8 vcnt instruction.
764
765 setOperationAction(ISD::CTLZ, MVT::v1i64, Expand);
766 setOperationAction(ISD::CTLZ, MVT::v2i64, Expand);
767
768 // NEON does not have single instruction CTTZ for vectors.
770 setOperationAction(ISD::CTTZ, MVT::v4i16, Custom);
771 setOperationAction(ISD::CTTZ, MVT::v2i32, Custom);
772 setOperationAction(ISD::CTTZ, MVT::v1i64, Custom);
773
774 setOperationAction(ISD::CTTZ, MVT::v16i8, Custom);
775 setOperationAction(ISD::CTTZ, MVT::v8i16, Custom);
776 setOperationAction(ISD::CTTZ, MVT::v4i32, Custom);
777 setOperationAction(ISD::CTTZ, MVT::v2i64, Custom);
778
783
788
792 }
793
794 // NEON only has FMA instructions as of VFP4.
795 if (!Subtarget->hasVFP4Base()) {
796 setOperationAction(ISD::FMA, MVT::v2f32, Expand);
797 setOperationAction(ISD::FMA, MVT::v4f32, Expand);
798 }
799
802
803 // It is legal to extload from v4i8 to v4i16 or v4i32.
804 for (MVT Ty : {MVT::v8i8, MVT::v4i8, MVT::v2i8, MVT::v4i16, MVT::v2i16,
805 MVT::v2i32}) {
810 }
811 }
812
813 for (auto VT : {MVT::v8i8, MVT::v4i16, MVT::v2i32, MVT::v16i8, MVT::v8i16,
814 MVT::v4i32}) {
819 }
820 }
821
822 if (Subtarget->hasNEON() || Subtarget->hasMVEIntegerOps()) {
829 }
830 if (Subtarget->hasMVEIntegerOps()) {
833 ISD::SETCC});
834 }
835 if (Subtarget->hasMVEFloatOps()) {
837 }
838
839 if (!Subtarget->hasFP64()) {
840 // When targeting a floating-point unit with only single-precision
841 // operations, f64 is legal for the few double-precision instructions which
842 // are present However, no double-precision operations other than moves,
843 // loads and stores are provided by the hardware.
880 }
881
882 // STRICT_(U/S)INT_TO_FP specifically use the input MVT to register with
883 // setOperationAction() as opposed to other opcodes that use the output MVT
884 // All inputs should be i32 due to type legalization
887
890
891 if (!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) {
894 if (Subtarget->hasFullFP16()) {
897 }
898 } else {
900 }
901
902 if (!Subtarget->hasFP16()) {
905 } else {
908 }
909
910 computeRegisterProperties(Subtarget->getRegisterInfo());
911
912 // ARM does not have floating-point extending loads.
913 for (MVT VT : MVT::fp_valuetypes()) {
914 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f32, Expand);
915 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f16, Expand);
916 setLoadExtAction(ISD::EXTLOAD, VT, MVT::bf16, Expand);
917 }
918
919 // ... or truncating stores
920 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
921 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
922 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
923 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
924 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
925
926 // ARM does not have i1 sign extending load.
927 for (MVT VT : MVT::integer_valuetypes())
929
930 // ARM supports all 4 flavors of integer indexed load / store.
931 if (!Subtarget->isThumb1Only()) {
932 for (unsigned im = (unsigned)ISD::PRE_INC;
934 setIndexedLoadAction(im, MVT::i1, Legal);
935 setIndexedLoadAction(im, MVT::i8, Legal);
936 setIndexedLoadAction(im, MVT::i16, Legal);
937 setIndexedLoadAction(im, MVT::i32, Legal);
938 setIndexedStoreAction(im, MVT::i1, Legal);
939 setIndexedStoreAction(im, MVT::i8, Legal);
940 setIndexedStoreAction(im, MVT::i16, Legal);
941 setIndexedStoreAction(im, MVT::i32, Legal);
942 }
943 } else {
944 // Thumb-1 has limited post-inc load/store support - LDM r0!, {r1}.
947 }
948
949 // Custom loads/stores to possible use __aeabi_uread/write*
950 if (TT.isTargetAEABI() && !Subtarget->allowsUnalignedMem()) {
955 }
956
961
962 if (!Subtarget->isThumb1Only()) {
965 }
966
971 if (Subtarget->hasDSP()) {
980 }
981 if (Subtarget->hasBaseDSP()) {
984 }
985
986 // i64 operation support.
989 if (Subtarget->isThumb1Only()) {
992 }
993 if (Subtarget->isThumb1Only() || !Subtarget->hasV6Ops()
994 || (Subtarget->isThumb2() && !Subtarget->hasDSP()))
996
1006
1007 // MVE lowers 64 bit shifts to lsll and lsrl
1008 // assuming that ISD::SRL and SRA of i64 are already marked custom
1009 if (Subtarget->hasMVEIntegerOps())
1011
1012 // Expand to __aeabi_l{lsl,lsr,asr} calls for Thumb1.
1013 if (Subtarget->isThumb1Only()) {
1017 }
1018
1019 if (!Subtarget->isThumb1Only() && Subtarget->hasV6T2Ops())
1021
1022 // ARM does not have ROTL.
1027 }
1029 // TODO: These two should be set to LibCall, but this currently breaks
1030 // the Linux kernel build. See #101786.
1033 if (!Subtarget->hasV5TOps() || Subtarget->isThumb1Only()) {
1036 }
1037
1038 // @llvm.readcyclecounter requires the Performance Monitors extension.
1039 // Default to the 0 expansion on unsupported platforms.
1040 // FIXME: Technically there are older ARM CPUs that have
1041 // implementation-specific ways of obtaining this information.
1042 if (Subtarget->hasPerfMon())
1044
1045 // Only ARMv6 has BSWAP.
1046 if (!Subtarget->hasV6Ops())
1048
1049 bool hasDivide = Subtarget->isThumb() ? Subtarget->hasDivideInThumbMode()
1050 : Subtarget->hasDivideInARMMode();
1051 if (!hasDivide) {
1052 // These are expanded into libcalls if the cpu doesn't have HW divider.
1055 }
1056
1057 if (TT.isOSWindows() && !Subtarget->hasDivideInThumbMode()) {
1060
1063 }
1064
1067
1068 // Register based DivRem for AEABI (RTABI 4.2)
1069 if (TT.isTargetAEABI() || TT.isAndroid() || TT.isTargetGNUAEABI() ||
1070 TT.isTargetMuslAEABI() || TT.isOSFuchsia() || TT.isOSWindows()) {
1073 HasStandaloneRem = false;
1074
1079 } else {
1082 }
1083
1088
1089 setOperationAction(ISD::TRAP, MVT::Other, Legal);
1091
1092 // Use the default implementation.
1094 setOperationAction(ISD::VAARG, MVT::Other, Expand);
1096 setOperationAction(ISD::VAEND, MVT::Other, Expand);
1099
1100 if (TT.isOSWindows())
1102 else
1104
1105 // ARMv6 Thumb1 (except for CPUs that support dmb / dsb) and earlier use
1106 // the default expansion.
1107 InsertFencesForAtomic = false;
1108 if (Subtarget->hasAnyDataBarrier() &&
1109 (!Subtarget->isThumb() || Subtarget->hasV8MBaselineOps())) {
1110 // ATOMIC_FENCE needs custom lowering; the others should have been expanded
1111 // to ldrex/strex loops already.
1113 if (!Subtarget->isThumb() || !Subtarget->isMClass())
1115
1116 // On v8, we have particularly efficient implementations of atomic fences
1117 // if they can be combined with nearby atomic loads and stores.
1118 if (!Subtarget->hasAcquireRelease() ||
1119 getTargetMachine().getOptLevel() == CodeGenOptLevel::None) {
1120 // Automatically insert fences (dmb ish) around ATOMIC_SWAP etc.
1121 InsertFencesForAtomic = true;
1122 }
1123 } else {
1124 // If there's anything we can use as a barrier, go through custom lowering
1125 // for ATOMIC_FENCE.
1126 // If target has DMB in thumb, Fences can be inserted.
1127 if (Subtarget->hasDataBarrier())
1128 InsertFencesForAtomic = true;
1129
1131 Subtarget->hasAnyDataBarrier() ? Custom : Expand);
1132
1133 // Set them all for libcall, which will force libcalls.
1146 // Mark ATOMIC_LOAD and ATOMIC_STORE custom so we can handle the
1147 // Unordered/Monotonic case.
1148 if (!InsertFencesForAtomic) {
1151 }
1152 }
1153
1154 // Compute supported atomic widths.
1155 if (TT.isOSLinux() || (!Subtarget->isMClass() && Subtarget->hasV6Ops())) {
1156 // For targets where __sync_* routines are reliably available, we use them
1157 // if necessary.
1158 //
1159 // ARM Linux always supports 64-bit atomics through kernel-assisted atomic
1160 // routines (kernel 3.1 or later). FIXME: Not with compiler-rt?
1161 //
1162 // ARMv6 targets have native instructions in ARM mode. For Thumb mode,
1163 // such targets should provide __sync_* routines, which use the ARM mode
1164 // instructions. (ARMv6 doesn't have dmb, but it has an equivalent
1165 // encoding; see ARMISD::MEMBARRIER_MCR.)
1167 } else if ((Subtarget->isMClass() && Subtarget->hasV8MBaselineOps()) ||
1168 Subtarget->hasForced32BitAtomics()) {
1169 // Cortex-M (besides Cortex-M0) have 32-bit atomics.
1171 } else {
1172 // We can't assume anything about other targets; just use libatomic
1173 // routines.
1175 }
1176
1178
1180
1181 // Requires SXTB/SXTH, available on v6 and up in both ARM and Thumb modes.
1182 if (!Subtarget->hasV6Ops()) {
1185 }
1187
1188 if (!Subtarget->useSoftFloat() && Subtarget->hasFPRegs() &&
1189 !Subtarget->isThumb1Only()) {
1190 // Turn f64->i64 into VMOVRRD, i64 -> f64 to VMOVDRR
1191 // iff target supports vfp2.
1201 }
1202
1203 // We want to custom lower some of our intrinsics.
1208
1218 if (Subtarget->hasFullFP16()) {
1222 }
1223
1225
1228 if (Subtarget->hasFullFP16())
1232 setOperationAction(ISD::BR_JT, MVT::Other, Custom);
1233
1234 // We don't support sin/cos/fmod/copysign/pow
1243 if (!Subtarget->useSoftFloat() && Subtarget->hasVFP2Base() &&
1244 !Subtarget->isThumb1Only()) {
1247 }
1250
1251 if (!Subtarget->hasVFP4Base()) {
1254 }
1255
1256 // Various VFP goodness
1257 if (!Subtarget->useSoftFloat() && !Subtarget->isThumb1Only()) {
1258 // FP-ARMv8 adds f64 <-> f16 conversion. Before that it should be expanded.
1259 if (!Subtarget->hasFPARMv8Base() || !Subtarget->hasFP64()) {
1264 }
1265
1266 // fp16 is a special v7 extension that adds f16 <-> f32 conversions.
1267 if (!Subtarget->hasFP16()) {
1272 }
1273
1274 // Strict floating-point comparisons need custom lowering.
1281 }
1282
1283 // FP-ARMv8 implements a lot of rounding-like FP operations.
1284 if (Subtarget->hasFPARMv8Base()) {
1285 for (auto Op :
1292 setOperationAction(Op, MVT::f32, Legal);
1293
1294 if (Subtarget->hasFP64())
1295 setOperationAction(Op, MVT::f64, Legal);
1296 }
1297
1298 if (Subtarget->hasNEON()) {
1303 }
1304 }
1305
1306 // FP16 often need to be promoted to call lib functions
1307 // clang-format off
1308 if (Subtarget->hasFullFP16()) {
1312
1313 for (auto Op : {ISD::FREM, ISD::FPOW, ISD::FPOWI,
1327 setOperationAction(Op, MVT::f16, Promote);
1328 }
1329
1330 // Round-to-integer need custom lowering for fp16, as Promote doesn't work
1331 // because the result type is integer.
1333 setOperationAction(Op, MVT::f16, Custom);
1334
1340 setOperationAction(Op, MVT::f16, Legal);
1341 }
1342 // clang-format on
1343 }
1344
1345 if (Subtarget->hasNEON()) {
1346 // vmin and vmax aren't available in a scalar form, so we can use
1347 // a NEON instruction with an undef lane instead.
1356
1357 if (Subtarget->hasV8Ops()) {
1362 setOperationAction(Op, MVT::v2f32, Legal);
1363 setOperationAction(Op, MVT::v4f32, Legal);
1364 }
1365 }
1366
1367 if (Subtarget->hasFullFP16()) {
1372
1377
1382 setOperationAction(Op, MVT::v4f16, Legal);
1383 setOperationAction(Op, MVT::v8f16, Legal);
1384 }
1385 }
1386 }
1387
1388 // On MSVC, both 32-bit and 64-bit, ldexpf(f32) is not defined. MinGW has
1389 // it, but it's just a wrapper around ldexp.
1390 if (TT.isOSWindows()) {
1392 if (isOperationExpand(Op, MVT::f32))
1393 setOperationAction(Op, MVT::f32, Promote);
1394 }
1395
1396 // LegalizeDAG currently can't expand fp16 LDEXP/FREXP on targets where i16
1397 // isn't legal.
1399 if (isOperationExpand(Op, MVT::f16))
1400 setOperationAction(Op, MVT::f16, Promote);
1401
1402 // We have target-specific dag combine patterns for the following nodes:
1403 // ARMISD::VMOVRRD - No need to call setTargetDAGCombine
1406
1407 if (Subtarget->hasMVEIntegerOps())
1409
1410 if (Subtarget->hasV6Ops())
1412 if (Subtarget->isThumb1Only())
1414 // Attempt to lower smin/smax to ssat/usat
1415 if ((!Subtarget->isThumb() && Subtarget->hasV6Ops()) ||
1416 Subtarget->isThumb2()) {
1418 }
1419
1421
1422 if (Subtarget->useSoftFloat() || Subtarget->isThumb1Only() ||
1423 !Subtarget->hasVFP2Base() || Subtarget->hasMinSize())
1425 else
1427
1428 //// temporary - rewrite interface to use type
1431 MaxStoresPerMemcpy = 4; // For @llvm.memcpy -> sequence of stores
1433 MaxStoresPerMemmove = 4; // For @llvm.memmove -> sequence of stores
1435
1436 // On ARM arguments smaller than 4 bytes are extended, so all arguments
1437 // are at least 4 bytes aligned.
1439
1440 // Prefer likely predicted branches to selects on out-of-order cores.
1441 PredictableSelectIsExpensive = Subtarget->getSchedModel().isOutOfOrder();
1442
1443 setPrefLoopAlignment(Align(1ULL << Subtarget->getPreferBranchLogAlignment()));
1445 Align(1ULL << Subtarget->getPreferBranchLogAlignment()));
1446
1447 setMinFunctionAlignment(Subtarget->isThumb() ? Align(2) : Align(4));
1448
1449 IsStrictFPEnabled = true;
1450}
1451
1453 return Subtarget->useSoftFloat();
1454}
1455
1457 return !Subtarget->isThumb1Only() && VT.getSizeInBits() <= 32;
1458}
1459
1460// FIXME: It might make sense to define the representative register class as the
1461// nearest super-register that has a non-null superset. For example, DPR_VFP2 is
1462// a super-register of SPR, and DPR is a superset if DPR_VFP2. Consequently,
1463// SPR's representative would be DPR_VFP2. This should work well if register
1464// pressure tracking were modified such that a register use would increment the
1465// pressure of the register class's representative and all of it's super
1466// classes' representatives transitively. We have not implemented this because
1467// of the difficulty prior to coalescing of modeling operand register classes
1468// due to the common occurrence of cross class copies and subregister insertions
1469// and extractions.
1470std::pair<const TargetRegisterClass *, uint8_t>
1472 MVT VT) const {
1473 const TargetRegisterClass *RRC = nullptr;
1474 uint8_t Cost = 1;
1475 switch (VT.SimpleTy) {
1476 default:
1478 // Use DPR as representative register class for all floating point
1479 // and vector types. Since there are 32 SPR registers and 32 DPR registers so
1480 // the cost is 1 for both f32 and f64.
1481 case MVT::f32: case MVT::f64: case MVT::v8i8: case MVT::v4i16:
1482 case MVT::v2i32: case MVT::v1i64: case MVT::v2f32:
1483 RRC = &ARM::DPRRegClass;
1484 // When NEON is used for SP, only half of the register file is available
1485 // because operations that define both SP and DP results will be constrained
1486 // to the VFP2 class (D0-D15). We currently model this constraint prior to
1487 // coalescing by double-counting the SP regs. See the FIXME above.
1488 if (Subtarget->useNEONForSinglePrecisionFP())
1489 Cost = 2;
1490 break;
1491 case MVT::v16i8: case MVT::v8i16: case MVT::v4i32: case MVT::v2i64:
1492 case MVT::v4f32: case MVT::v2f64:
1493 RRC = &ARM::DPRRegClass;
1494 Cost = 2;
1495 break;
1496 case MVT::v4i64:
1497 RRC = &ARM::DPRRegClass;
1498 Cost = 4;
1499 break;
1500 case MVT::v8i64:
1501 RRC = &ARM::DPRRegClass;
1502 Cost = 8;
1503 break;
1504 }
1505 return std::make_pair(RRC, Cost);
1506}
1507
1509 EVT VT) const {
1510 if (!VT.isVector())
1511 return getPointerTy(DL);
1512
1513 // MVE has a predicate register.
1514 if (Subtarget->hasMVEIntegerOps())
1515 return EVT::getVectorVT(C, MVT::i1, VT.getVectorElementCount());
1516
1518}
1519
1520/// getRegClassFor - Return the register class that should be used for the
1521/// specified value type.
1522const TargetRegisterClass *
1523ARMTargetLowering::getRegClassFor(MVT VT, bool isDivergent) const {
1524 (void)isDivergent;
1525 // Map v4i64 to QQ registers but do not make the type legal. Similarly map
1526 // v8i64 to QQQQ registers. v4i64 and v8i64 are only used for REG_SEQUENCE to
1527 // load / store 4 to 8 consecutive NEON D registers, or 2 to 4 consecutive
1528 // MVE Q registers.
1529 if (Subtarget->hasNEON()) {
1530 if (VT == MVT::v4i64)
1531 return &ARM::QQPRRegClass;
1532 if (VT == MVT::v8i64)
1533 return &ARM::QQQQPRRegClass;
1534 }
1535 if (Subtarget->hasMVEIntegerOps()) {
1536 if (VT == MVT::v4i64)
1537 return &ARM::MQQPRRegClass;
1538 if (VT == MVT::v8i64)
1539 return &ARM::MQQQQPRRegClass;
1540 }
1542}
1543
1544// memcpy, and other memory intrinsics, typically tries to use LDM/STM if the
1545// source/dest is aligned and the copy size is large enough. We therefore want
1546// to align such objects passed to memory intrinsics.
1548 Align &PrefAlign) const {
1549 if (!isa<MemIntrinsic>(CI))
1550 return false;
1551 MinSize = 8;
1552 // On ARM11 onwards (excluding M class) 8-byte aligned LDM is typically 1
1553 // cycle faster than 4-byte aligned LDM.
1554 PrefAlign =
1555 (Subtarget->hasV6Ops() && !Subtarget->isMClass() ? Align(8) : Align(4));
1556 return true;
1557}
1558
1559// Create a fast isel object.
1561 FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo,
1562 const LibcallLoweringInfo *libcallLowering) const {
1563 return ARM::createFastISel(funcInfo, libInfo, libcallLowering);
1564}
1565
1567 unsigned NumVals = N->getNumValues();
1568 if (!NumVals)
1569 return Sched::RegPressure;
1570
1571 for (unsigned i = 0; i != NumVals; ++i) {
1572 EVT VT = N->getValueType(i);
1573 if (VT == MVT::Glue || VT == MVT::Other)
1574 continue;
1575 if (VT.isFloatingPoint() || VT.isVector())
1576 return Sched::ILP;
1577 }
1578
1579 if (!N->isMachineOpcode())
1580 return Sched::RegPressure;
1581
1582 // Load are scheduled for latency even if there instruction itinerary
1583 // is not available.
1584 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
1585 const MCInstrDesc &MCID = TII->get(N->getMachineOpcode());
1586
1587 if (MCID.getNumDefs() == 0)
1588 return Sched::RegPressure;
1589 if (!Itins->isEmpty() &&
1590 Itins->getOperandCycle(MCID.getSchedClass(), 0) > 2U)
1591 return Sched::ILP;
1592
1593 return Sched::RegPressure;
1594}
1595
1596//===----------------------------------------------------------------------===//
1597// Lowering Code
1598//===----------------------------------------------------------------------===//
1599
1600static bool isSRL16(const SDValue &Op) {
1601 if (Op.getOpcode() != ISD::SRL)
1602 return false;
1603 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
1604 return Const->getZExtValue() == 16;
1605 return false;
1606}
1607
1608static bool isSRA16(const SDValue &Op) {
1609 if (Op.getOpcode() != ISD::SRA)
1610 return false;
1611 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
1612 return Const->getZExtValue() == 16;
1613 return false;
1614}
1615
1616static bool isSHL16(const SDValue &Op) {
1617 if (Op.getOpcode() != ISD::SHL)
1618 return false;
1619 if (auto Const = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
1620 return Const->getZExtValue() == 16;
1621 return false;
1622}
1623
1624// Check for a signed 16-bit value. We special case SRA because it makes it
1625// more simple when also looking for SRAs that aren't sign extending a
1626// smaller value. Without the check, we'd need to take extra care with
1627// checking order for some operations.
1628static bool isS16(const SDValue &Op, SelectionDAG &DAG) {
1629 if (isSRA16(Op))
1630 return isSHL16(Op.getOperand(0));
1631 return DAG.ComputeNumSignBits(Op) == 17;
1632}
1633
1634/// IntCCToARMCC - Convert a DAG integer condition code to an ARM CC
1636 switch (CC) {
1637 default: llvm_unreachable("Unknown condition code!");
1638 case ISD::SETNE: return ARMCC::NE;
1639 case ISD::SETEQ: return ARMCC::EQ;
1640 case ISD::SETGT: return ARMCC::GT;
1641 case ISD::SETGE: return ARMCC::GE;
1642 case ISD::SETLT: return ARMCC::LT;
1643 case ISD::SETLE: return ARMCC::LE;
1644 case ISD::SETUGT: return ARMCC::HI;
1645 case ISD::SETUGE: return ARMCC::HS;
1646 case ISD::SETULT: return ARMCC::LO;
1647 case ISD::SETULE: return ARMCC::LS;
1648 }
1649}
1650
1651/// FPCCToARMCC - Convert a DAG fp condition code to an ARM CC.
1653 ARMCC::CondCodes &CondCode2) {
1654 CondCode2 = ARMCC::AL;
1655 switch (CC) {
1656 default: llvm_unreachable("Unknown FP condition!");
1657 case ISD::SETEQ:
1658 case ISD::SETOEQ: CondCode = ARMCC::EQ; break;
1659 case ISD::SETGT:
1660 case ISD::SETOGT: CondCode = ARMCC::GT; break;
1661 case ISD::SETGE:
1662 case ISD::SETOGE: CondCode = ARMCC::GE; break;
1663 case ISD::SETOLT: CondCode = ARMCC::MI; break;
1664 case ISD::SETOLE: CondCode = ARMCC::LS; break;
1665 case ISD::SETONE: CondCode = ARMCC::MI; CondCode2 = ARMCC::GT; break;
1666 case ISD::SETO: CondCode = ARMCC::VC; break;
1667 case ISD::SETUO: CondCode = ARMCC::VS; break;
1668 case ISD::SETUEQ: CondCode = ARMCC::EQ; CondCode2 = ARMCC::VS; break;
1669 case ISD::SETUGT: CondCode = ARMCC::HI; break;
1670 case ISD::SETUGE: CondCode = ARMCC::PL; break;
1671 case ISD::SETLT:
1672 case ISD::SETULT: CondCode = ARMCC::LT; break;
1673 case ISD::SETLE:
1674 case ISD::SETULE: CondCode = ARMCC::LE; break;
1675 case ISD::SETNE:
1676 case ISD::SETUNE: CondCode = ARMCC::NE; break;
1677 }
1678}
1679
1680//===----------------------------------------------------------------------===//
1681// Calling Convention Implementation
1682//===----------------------------------------------------------------------===//
1683
1684/// getEffectiveCallingConv - Get the effective calling convention, taking into
1685/// account presence of floating point hardware and calling convention
1686/// limitations, such as support for variadic functions.
1689 bool isVarArg) const {
1690 switch (CC) {
1691 default:
1692 // Unknown CCs are rejected when calling convention lowering is required.
1695 case CallingConv::GHC:
1697 return CC;
1703 case CallingConv::Swift:
1706 case CallingConv::C:
1707 case CallingConv::Tail:
1708 if (!Subtarget->isAAPCS_ABI())
1709 return CallingConv::ARM_APCS;
1710 else if (Subtarget->isTargetHardFloat() && !isVarArg)
1712 else
1714 case CallingConv::Fast:
1716 if (!Subtarget->isAAPCS_ABI()) {
1717 if (Subtarget->hasFPRegs() && !Subtarget->isThumb1Only() && !isVarArg)
1718 return CallingConv::Fast;
1719 return CallingConv::ARM_APCS;
1720 } else if (Subtarget->hasFPRegs() && !Subtarget->isThumb1Only() &&
1721 !isVarArg)
1723 else
1725 }
1726}
1727
1729 bool isVarArg) const {
1730 return CCAssignFnForNode(CC, false, isVarArg);
1731}
1732
1734 bool isVarArg) const {
1735 return CCAssignFnForNode(CC, true, isVarArg);
1736}
1737
1738/// CCAssignFnForNode - Selects the correct CCAssignFn for the given
1739/// CallingConvention.
1740CCAssignFn *ARMTargetLowering::CCAssignFnForNode(CallingConv::ID CC,
1741 bool Return,
1742 bool isVarArg) const {
1743 switch (getEffectiveCallingConv(CC, isVarArg)) {
1744 default:
1745 report_fatal_error("Unsupported calling convention");
1747 return (Return ? RetCC_ARM_APCS : CC_ARM_APCS);
1749 return (Return ? RetCC_ARM_AAPCS : CC_ARM_AAPCS);
1751 return (Return ? RetCC_ARM_AAPCS_VFP : CC_ARM_AAPCS_VFP);
1752 case CallingConv::Fast:
1753 return (Return ? RetFastCC_ARM_APCS : FastCC_ARM_APCS);
1754 case CallingConv::GHC:
1755 return (Return ? RetCC_ARM_APCS : CC_ARM_APCS_GHC);
1757 return (Return ? RetCC_ARM_AAPCS : CC_ARM_AAPCS);
1759 return (Return ? RetCC_ARM_AAPCS : CC_ARM_AAPCS);
1761 return (Return ? RetCC_ARM_AAPCS : CC_ARM_Win32_CFGuard_Check);
1762 }
1763}
1764
1765SDValue ARMTargetLowering::MoveToHPR(const SDLoc &dl, SelectionDAG &DAG,
1766 MVT LocVT, MVT ValVT, SDValue Val) const {
1767 Val = DAG.getNode(ISD::BITCAST, dl, MVT::getIntegerVT(LocVT.getSizeInBits()),
1768 Val);
1769 if (Subtarget->hasFullFP16()) {
1770 Val = DAG.getNode(ARMISD::VMOVhr, dl, ValVT, Val);
1771 } else {
1772 Val = DAG.getNode(ISD::TRUNCATE, dl,
1773 MVT::getIntegerVT(ValVT.getSizeInBits()), Val);
1774 Val = DAG.getNode(ISD::BITCAST, dl, ValVT, Val);
1775 }
1776 return Val;
1777}
1778
1779SDValue ARMTargetLowering::MoveFromHPR(const SDLoc &dl, SelectionDAG &DAG,
1780 MVT LocVT, MVT ValVT,
1781 SDValue Val) const {
1782 if (Subtarget->hasFullFP16()) {
1783 Val = DAG.getNode(ARMISD::VMOVrh, dl,
1784 MVT::getIntegerVT(LocVT.getSizeInBits()), Val);
1785 } else {
1786 Val = DAG.getNode(ISD::BITCAST, dl,
1787 MVT::getIntegerVT(ValVT.getSizeInBits()), Val);
1788 Val = DAG.getNode(ISD::ZERO_EXTEND, dl,
1789 MVT::getIntegerVT(LocVT.getSizeInBits()), Val);
1790 }
1791 return DAG.getNode(ISD::BITCAST, dl, LocVT, Val);
1792}
1793
1794/// LowerCallResult - Lower the result values of a call into the
1795/// appropriate copies out of appropriate physical registers.
1796SDValue ARMTargetLowering::LowerCallResult(
1797 SDValue Chain, SDValue InGlue, CallingConv::ID CallConv, bool isVarArg,
1798 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl,
1799 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals, bool isThisReturn,
1800 SDValue ThisVal, bool isCmseNSCall) const {
1801 // Assign locations to each value returned by this call.
1803 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs,
1804 *DAG.getContext());
1805 CCInfo.AnalyzeCallResult(Ins, CCAssignFnForReturn(CallConv, isVarArg));
1806
1807 // Copy all of the result registers out of their specified physreg.
1808 for (unsigned i = 0; i != RVLocs.size(); ++i) {
1809 CCValAssign VA = RVLocs[i];
1810
1811 // Pass 'this' value directly from the argument to return value, to avoid
1812 // reg unit interference
1813 if (i == 0 && isThisReturn) {
1814 assert(!VA.needsCustom() && VA.getLocVT() == MVT::i32 &&
1815 "unexpected return calling convention register assignment");
1816 InVals.push_back(ThisVal);
1817 continue;
1818 }
1819
1820 SDValue Val;
1821 if (VA.needsCustom() &&
1822 (VA.getLocVT() == MVT::f64 || VA.getLocVT() == MVT::v2f64)) {
1823 // Handle f64 or half of a v2f64.
1824 SDValue Lo = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32,
1825 InGlue);
1826 Chain = Lo.getValue(1);
1827 InGlue = Lo.getValue(2);
1828 VA = RVLocs[++i]; // skip ahead to next loc
1829 SDValue Hi = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32,
1830 InGlue);
1831 Chain = Hi.getValue(1);
1832 InGlue = Hi.getValue(2);
1833 if (!Subtarget->isLittle())
1834 std::swap (Lo, Hi);
1835 Val = DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi);
1836
1837 if (VA.getLocVT() == MVT::v2f64) {
1838 SDValue Vec = DAG.getNode(ISD::UNDEF, dl, MVT::v2f64);
1839 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Vec, Val,
1840 DAG.getConstant(0, dl, MVT::i32));
1841
1842 VA = RVLocs[++i]; // skip ahead to next loc
1843 Lo = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, InGlue);
1844 Chain = Lo.getValue(1);
1845 InGlue = Lo.getValue(2);
1846 VA = RVLocs[++i]; // skip ahead to next loc
1847 Hi = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), MVT::i32, InGlue);
1848 Chain = Hi.getValue(1);
1849 InGlue = Hi.getValue(2);
1850 if (!Subtarget->isLittle())
1851 std::swap (Lo, Hi);
1852 Val = DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi);
1853 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Vec, Val,
1854 DAG.getConstant(1, dl, MVT::i32));
1855 }
1856 } else {
1857 Val = DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), VA.getLocVT(),
1858 InGlue);
1859 Chain = Val.getValue(1);
1860 InGlue = Val.getValue(2);
1861 }
1862
1863 switch (VA.getLocInfo()) {
1864 default: llvm_unreachable("Unknown loc info!");
1865 case CCValAssign::Full: break;
1866 case CCValAssign::BCvt:
1867 Val = DAG.getNode(ISD::BITCAST, dl, VA.getValVT(), Val);
1868 break;
1869 }
1870
1871 // f16 arguments have their size extended to 4 bytes and passed as if they
1872 // had been copied to the LSBs of a 32-bit register.
1873 // For that, it's passed extended to i32 (soft ABI) or to f32 (hard ABI)
1874 if (VA.needsCustom() &&
1875 (VA.getValVT() == MVT::f16 || VA.getValVT() == MVT::bf16))
1876 Val = MoveToHPR(dl, DAG, VA.getLocVT(), VA.getValVT(), Val);
1877
1878 // On CMSE Non-secure Calls, call results (returned values) whose bitwidth
1879 // is less than 32 bits must be sign- or zero-extended after the call for
1880 // security reasons. Although the ABI mandates an extension done by the
1881 // callee, the latter cannot be trusted to follow the rules of the ABI.
1882 const ISD::InputArg &Arg = Ins[VA.getValNo()];
1883 if (isCmseNSCall && Arg.ArgVT.isScalarInteger() &&
1884 VA.getLocVT().isScalarInteger() && Arg.ArgVT.bitsLT(MVT::i32))
1885 Val = handleCMSEValue(Val, Arg, DAG, dl);
1886
1887 InVals.push_back(Val);
1888 }
1889
1890 return Chain;
1891}
1892
1893std::pair<SDValue, MachinePointerInfo> ARMTargetLowering::computeAddrForCallArg(
1894 const SDLoc &dl, SelectionDAG &DAG, const CCValAssign &VA, SDValue StackPtr,
1895 bool IsTailCall, int SPDiff) const {
1896 SDValue DstAddr;
1897 MachinePointerInfo DstInfo;
1898 int32_t Offset = VA.getLocMemOffset();
1900
1901 if (IsTailCall) {
1902 Offset += SPDiff;
1903 auto PtrVT = getPointerTy(DAG.getDataLayout());
1904 int Size = VA.getLocVT().getFixedSizeInBits() / 8;
1905 int FI = MF.getFrameInfo().CreateFixedObject(Size, Offset, true);
1906 DstAddr = DAG.getFrameIndex(FI, PtrVT);
1907 DstInfo =
1909 } else {
1910 SDValue PtrOff = DAG.getIntPtrConstant(Offset, dl);
1911 DstAddr = DAG.getNode(ISD::ADD, dl, getPointerTy(DAG.getDataLayout()),
1912 StackPtr, PtrOff);
1913 DstInfo =
1915 }
1916
1917 return std::make_pair(DstAddr, DstInfo);
1918}
1919
1920// Returns the type of copying which is required to set up a byval argument to
1921// a tail-called function. This isn't needed for non-tail calls, because they
1922// always need the equivalent of CopyOnce, but tail-calls sometimes need two to
1923// avoid clobbering another argument (CopyViaTemp), and sometimes can be
1924// optimised to zero copies when forwarding an argument from the caller's
1925// caller (NoCopy).
1926ARMTargetLowering::ByValCopyKind ARMTargetLowering::ByValNeedsCopyForTailCall(
1927 SelectionDAG &DAG, SDValue Src, SDValue Dst, ISD::ArgFlagsTy Flags) const {
1928 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
1929 ARMFunctionInfo *AFI = DAG.getMachineFunction().getInfo<ARMFunctionInfo>();
1930
1931 // Globals are always safe to copy from.
1933 return CopyOnce;
1934
1935 // Can only analyse frame index nodes, conservatively assume we need a
1936 // temporary.
1937 auto *SrcFrameIdxNode = dyn_cast<FrameIndexSDNode>(Src);
1938 auto *DstFrameIdxNode = dyn_cast<FrameIndexSDNode>(Dst);
1939 if (!SrcFrameIdxNode || !DstFrameIdxNode)
1940 return CopyViaTemp;
1941
1942 int SrcFI = SrcFrameIdxNode->getIndex();
1943 int DstFI = DstFrameIdxNode->getIndex();
1944 assert(MFI.isFixedObjectIndex(DstFI) &&
1945 "byval passed in non-fixed stack slot");
1946
1947 int64_t SrcOffset = MFI.getObjectOffset(SrcFI);
1948 int64_t DstOffset = MFI.getObjectOffset(DstFI);
1949
1950 // If the source is in the local frame, then the copy to the argument memory
1951 // is always valid.
1952 bool FixedSrc = MFI.isFixedObjectIndex(SrcFI);
1953 if (!FixedSrc ||
1954 (FixedSrc && SrcOffset < -(int64_t)AFI->getArgRegsSaveSize()))
1955 return CopyOnce;
1956
1957 // In the case of byval arguments split between registers and the stack,
1958 // computeAddrForCallArg returns a FrameIndex which corresponds only to the
1959 // stack portion, but the Src SDValue will refer to the full value, including
1960 // the local stack memory that the register portion gets stored into. We only
1961 // need to compare them for equality, so normalise on the full value version.
1962 uint64_t RegSize = Flags.getByValSize() - MFI.getObjectSize(DstFI);
1963 DstOffset -= RegSize;
1964
1965 // If the value is already in the correct location, then no copying is
1966 // needed. If not, then we need to copy via a temporary.
1967 if (SrcOffset == DstOffset)
1968 return NoCopy;
1969 else
1970 return CopyViaTemp;
1971}
1972
1973void ARMTargetLowering::PassF64ArgInRegs(const SDLoc &dl, SelectionDAG &DAG,
1974 SDValue Chain, SDValue &Arg,
1975 RegsToPassVector &RegsToPass,
1976 CCValAssign &VA, CCValAssign &NextVA,
1977 SDValue &StackPtr,
1978 SmallVectorImpl<SDValue> &MemOpChains,
1979 bool IsTailCall,
1980 int SPDiff) const {
1981 SDValue fmrrd = DAG.getNode(ARMISD::VMOVRRD, dl,
1982 DAG.getVTList(MVT::i32, MVT::i32), Arg);
1983 unsigned id = Subtarget->isLittle() ? 0 : 1;
1984 RegsToPass.push_back(std::make_pair(VA.getLocReg(), fmrrd.getValue(id)));
1985
1986 if (NextVA.isRegLoc())
1987 RegsToPass.push_back(std::make_pair(NextVA.getLocReg(), fmrrd.getValue(1-id)));
1988 else {
1989 assert(NextVA.isMemLoc());
1990 if (!StackPtr.getNode())
1991 StackPtr = DAG.getCopyFromReg(Chain, dl, ARM::SP,
1993
1994 SDValue DstAddr;
1995 MachinePointerInfo DstInfo;
1996 std::tie(DstAddr, DstInfo) =
1997 computeAddrForCallArg(dl, DAG, NextVA, StackPtr, IsTailCall, SPDiff);
1998 MemOpChains.push_back(
1999 DAG.getStore(Chain, dl, fmrrd.getValue(1 - id), DstAddr, DstInfo));
2000 }
2001}
2002
2003static bool canGuaranteeTCO(CallingConv::ID CC, bool GuaranteeTailCalls) {
2004 return (CC == CallingConv::Fast && GuaranteeTailCalls) ||
2006}
2007
2008/// LowerCall - Lowering a call into a callseq_start <-
2009/// ARMISD:CALL <- callseq_end chain. Also add input and output parameter
2010/// nodes.
2011SDValue
2012ARMTargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI,
2013 SmallVectorImpl<SDValue> &InVals) const {
2014 SelectionDAG &DAG = CLI.DAG;
2015 SDLoc &dl = CLI.DL;
2016 SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
2017 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2018 SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
2019 SDValue Chain = CLI.Chain;
2020 SDValue Callee = CLI.Callee;
2021 bool &isTailCall = CLI.IsTailCall;
2022 CallingConv::ID CallConv = CLI.CallConv;
2023 bool doesNotRet = CLI.DoesNotReturn;
2024 bool isVarArg = CLI.IsVarArg;
2025 const CallBase *CB = CLI.CB;
2026
2028 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
2029 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
2030 MachineFunction::CallSiteInfo CSInfo;
2031 bool isStructRet = (Outs.empty()) ? false : Outs[0].Flags.isSRet();
2032 bool isThisReturn = false;
2033 bool isCmseNSCall = false;
2034 bool isSibCall = false;
2035 bool PreferIndirect = false;
2036 bool GuardWithBTI = false;
2037
2038 // Analyze operands of the call, assigning locations to each operand.
2040 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs,
2041 *DAG.getContext());
2042 CCInfo.AnalyzeCallOperands(Outs, CCAssignFnForCall(CallConv, isVarArg));
2043
2044 // Lower 'returns_twice' calls to a pseudo-instruction.
2045 if (CLI.CB && CLI.CB->getAttributes().hasFnAttr(Attribute::ReturnsTwice) &&
2046 !Subtarget->noBTIAtReturnTwice())
2047 GuardWithBTI = AFI->branchTargetEnforcement();
2048
2049 // Set type id for call site info.
2050 setTypeIdForCallsiteInfo(CB, MF, CSInfo);
2051
2052 // Determine whether this is a non-secure function call.
2053 if (CLI.CB && CLI.CB->getAttributes().hasFnAttr("cmse_nonsecure_call"))
2054 isCmseNSCall = true;
2055
2056 // Disable tail calls if they're not supported.
2057 if (!Subtarget->supportsTailCall())
2058 isTailCall = false;
2059
2060 // For both the non-secure calls and the returns from a CMSE entry function,
2061 // the function needs to do some extra work after the call, or before the
2062 // return, respectively, thus it cannot end with a tail call
2063 if (isCmseNSCall || AFI->isCmseNSEntryFunction())
2064 isTailCall = false;
2065
2066 if (isa<GlobalAddressSDNode>(Callee)) {
2067 // If we're optimizing for minimum size and the function is called three or
2068 // more times in this block, we can improve codesize by calling indirectly
2069 // as BLXr has a 16-bit encoding.
2070 auto *GV = cast<GlobalAddressSDNode>(Callee)->getGlobal();
2071 if (CLI.CB) {
2072 auto *BB = CLI.CB->getParent();
2073 PreferIndirect = Subtarget->isThumb() && Subtarget->hasMinSize() &&
2074 count_if(GV->users(), [&BB](const User *U) {
2075 return isa<Instruction>(U) &&
2076 cast<Instruction>(U)->getParent() == BB;
2077 }) > 2;
2078 }
2079 }
2080 if (isTailCall) {
2081 // Check if it's really possible to do a tail call.
2082 isTailCall =
2083 IsEligibleForTailCallOptimization(CLI, CCInfo, ArgLocs, PreferIndirect);
2084
2085 if (isTailCall && !getTargetMachine().Options.GuaranteedTailCallOpt &&
2086 CallConv != CallingConv::Tail && CallConv != CallingConv::SwiftTail)
2087 isSibCall = true;
2088
2089 // We don't support GuaranteedTailCallOpt for ARM, only automatically
2090 // detected sibcalls.
2091 if (isTailCall)
2092 ++NumTailCalls;
2093 }
2094
2095 if (!isTailCall && CLI.CB && CLI.CB->isMustTailCall())
2096 report_fatal_error("failed to perform tail call elimination on a call "
2097 "site marked musttail");
2098
2099 // Get a count of how many bytes are to be pushed on the stack.
2100 unsigned NumBytes = CCInfo.getStackSize();
2101
2102 // SPDiff is the byte offset of the call's argument area from the callee's.
2103 // Stores to callee stack arguments will be placed in FixedStackSlots offset
2104 // by this amount for a tail call. In a sibling call it must be 0 because the
2105 // caller will deallocate the entire stack and the callee still expects its
2106 // arguments to begin at SP+0. Completely unused for non-tail calls.
2107 int SPDiff = 0;
2108
2109 if (isTailCall && !isSibCall) {
2110 auto FuncInfo = MF.getInfo<ARMFunctionInfo>();
2111 unsigned NumReusableBytes = FuncInfo->getArgumentStackSize();
2112
2113 // Since callee will pop argument stack as a tail call, we must keep the
2114 // popped size 16-byte aligned.
2115 MaybeAlign StackAlign = DAG.getDataLayout().getStackAlignment();
2116 assert(StackAlign && "data layout string is missing stack alignment");
2117 NumBytes = alignTo(NumBytes, *StackAlign);
2118
2119 // SPDiff will be negative if this tail call requires more space than we
2120 // would automatically have in our incoming argument space. Positive if we
2121 // can actually shrink the stack.
2122 SPDiff = NumReusableBytes - NumBytes;
2123
2124 // If this call requires more stack than we have available from
2125 // LowerFormalArguments, tell FrameLowering to reserve space for it.
2126 if (SPDiff < 0 && AFI->getArgRegsSaveSize() < (unsigned)-SPDiff)
2127 AFI->setArgRegsSaveSize(-SPDiff);
2128 }
2129
2130 if (isSibCall) {
2131 // For sibling tail calls, memory operands are available in our caller's stack.
2132 NumBytes = 0;
2133 } else {
2134 // Adjust the stack pointer for the new arguments...
2135 // These operations are automatically eliminated by the prolog/epilog pass
2136 Chain = DAG.getCALLSEQ_START(Chain, isTailCall ? 0 : NumBytes, 0, dl);
2137 }
2138
2139 SDValue StackPtr =
2140 DAG.getCopyFromReg(Chain, dl, ARM::SP, getPointerTy(DAG.getDataLayout()));
2141
2142 RegsToPassVector RegsToPass;
2143 SmallVector<SDValue, 8> MemOpChains;
2144
2145 // If we are doing a tail-call, any byval arguments will be written to stack
2146 // space which was used for incoming arguments. If any the values being used
2147 // are incoming byval arguments to this function, then they might be
2148 // overwritten by the stores of the outgoing arguments. To avoid this, we
2149 // need to make a temporary copy of them in local stack space, then copy back
2150 // to the argument area.
2151 DenseMap<unsigned, SDValue> ByValTemporaries;
2152 SDValue ByValTempChain;
2153 if (isTailCall) {
2154 SmallVector<SDValue, 8> ByValCopyChains;
2155 for (const CCValAssign &VA : ArgLocs) {
2156 unsigned ArgIdx = VA.getValNo();
2157 SDValue Src = OutVals[ArgIdx];
2158 ISD::ArgFlagsTy Flags = Outs[ArgIdx].Flags;
2159
2160 if (!Flags.isByVal())
2161 continue;
2162
2163 SDValue Dst;
2164 MachinePointerInfo DstInfo;
2165 std::tie(Dst, DstInfo) =
2166 computeAddrForCallArg(dl, DAG, VA, SDValue(), true, SPDiff);
2167 ByValCopyKind Copy = ByValNeedsCopyForTailCall(DAG, Src, Dst, Flags);
2168
2169 if (Copy == NoCopy) {
2170 // If the argument is already at the correct offset on the stack
2171 // (because we are forwarding a byval argument from our caller), we
2172 // don't need any copying.
2173 continue;
2174 } else if (Copy == CopyOnce) {
2175 // If the argument is in our local stack frame, no other argument
2176 // preparation can clobber it, so we can copy it to the final location
2177 // later.
2178 ByValTemporaries[ArgIdx] = Src;
2179 } else {
2180 assert(Copy == CopyViaTemp && "unexpected enum value");
2181 // If we might be copying this argument from the outgoing argument
2182 // stack area, we need to copy via a temporary in the local stack
2183 // frame.
2184 int TempFrameIdx = MFI.CreateStackObject(
2185 Flags.getByValSize(), Flags.getNonZeroByValAlign(), false);
2186 SDValue Temp =
2187 DAG.getFrameIndex(TempFrameIdx, getPointerTy(DAG.getDataLayout()));
2188
2189 SDValue SizeNode = DAG.getConstant(Flags.getByValSize(), dl, MVT::i32);
2190 SDValue AlignNode =
2191 DAG.getConstant(Flags.getNonZeroByValAlign().value(), dl, MVT::i32);
2192
2193 SDVTList VTs = DAG.getVTList(MVT::Other, MVT::Glue);
2194 SDValue Ops[] = {Chain, Temp, Src, SizeNode, AlignNode};
2195 ByValCopyChains.push_back(
2196 DAG.getNode(ARMISD::COPY_STRUCT_BYVAL, dl, VTs, Ops));
2197 ByValTemporaries[ArgIdx] = Temp;
2198 }
2199 }
2200 if (!ByValCopyChains.empty())
2201 ByValTempChain =
2202 DAG.getNode(ISD::TokenFactor, dl, MVT::Other, ByValCopyChains);
2203 }
2204
2205 // During a tail call, stores to the argument area must happen after all of
2206 // the function's incoming arguments have been loaded because they may alias.
2207 // This is done by folding in a TokenFactor from LowerFormalArguments, but
2208 // there's no point in doing so repeatedly so this tracks whether that's
2209 // happened yet.
2210 bool AfterFormalArgLoads = false;
2211
2212 // Walk the register/memloc assignments, inserting copies/loads. In the case
2213 // of tail call optimization, arguments are handled later.
2214 for (unsigned i = 0, realArgIdx = 0, e = ArgLocs.size();
2215 i != e;
2216 ++i, ++realArgIdx) {
2217 CCValAssign &VA = ArgLocs[i];
2218 SDValue Arg = OutVals[realArgIdx];
2219 ISD::ArgFlagsTy Flags = Outs[realArgIdx].Flags;
2220 bool isByVal = Flags.isByVal();
2221
2222 // Promote the value if needed.
2223 switch (VA.getLocInfo()) {
2224 default: llvm_unreachable("Unknown loc info!");
2225 case CCValAssign::Full: break;
2226 case CCValAssign::SExt:
2227 Arg = DAG.getNode(ISD::SIGN_EXTEND, dl, VA.getLocVT(), Arg);
2228 break;
2229 case CCValAssign::ZExt:
2230 Arg = DAG.getNode(ISD::ZERO_EXTEND, dl, VA.getLocVT(), Arg);
2231 break;
2232 case CCValAssign::AExt:
2233 Arg = DAG.getNode(ISD::ANY_EXTEND, dl, VA.getLocVT(), Arg);
2234 break;
2235 case CCValAssign::BCvt:
2236 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg);
2237 break;
2238 }
2239
2240 if (isTailCall && VA.isMemLoc() && !AfterFormalArgLoads) {
2241 Chain = DAG.getStackArgumentTokenFactor(Chain);
2242 if (ByValTempChain) {
2243 // In case of large byval copies, re-using the stackframe for tail-calls
2244 // can lead to overwriting incoming arguments on the stack. Force
2245 // loading these stack arguments before the copy to avoid that.
2246 SmallVector<SDValue, 8> IncomingLoad;
2247 for (unsigned I = 0; I < OutVals.size(); ++I) {
2248 if (Outs[I].Flags.isByVal())
2249 continue;
2250
2251 SDValue OutVal = OutVals[I];
2252 LoadSDNode *OutLN = dyn_cast_or_null<LoadSDNode>(OutVal);
2253 if (!OutLN)
2254 continue;
2255
2256 FrameIndexSDNode *FIN =
2258 if (!FIN)
2259 continue;
2260
2261 if (!MFI.isFixedObjectIndex(FIN->getIndex()))
2262 continue;
2263
2264 for (const CCValAssign &VA : ArgLocs) {
2265 if (VA.isMemLoc())
2266 IncomingLoad.push_back(OutVal.getValue(1));
2267 }
2268 }
2269
2270 // Update the chain to force loads for potentially clobbered argument
2271 // loads to happen before the byval copy.
2272 if (!IncomingLoad.empty()) {
2273 IncomingLoad.push_back(Chain);
2274 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, IncomingLoad);
2275 }
2276
2277 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, Chain,
2278 ByValTempChain);
2279 }
2280 AfterFormalArgLoads = true;
2281 }
2282
2283 // f16 arguments have their size extended to 4 bytes and passed as if they
2284 // had been copied to the LSBs of a 32-bit register.
2285 // For that, it's passed extended to i32 (soft ABI) or to f32 (hard ABI)
2286 if (VA.needsCustom() &&
2287 (VA.getValVT() == MVT::f16 || VA.getValVT() == MVT::bf16)) {
2288 Arg = MoveFromHPR(dl, DAG, VA.getLocVT(), VA.getValVT(), Arg);
2289 } else {
2290 // f16 arguments could have been extended prior to argument lowering.
2291 // Mask them arguments if this is a CMSE nonsecure call.
2292 auto ArgVT = Outs[realArgIdx].ArgVT;
2293 if (isCmseNSCall && (ArgVT == MVT::f16)) {
2294 auto LocBits = VA.getLocVT().getSizeInBits();
2295 auto MaskValue = APInt::getLowBitsSet(LocBits, ArgVT.getSizeInBits());
2296 SDValue Mask =
2297 DAG.getConstant(MaskValue, dl, MVT::getIntegerVT(LocBits));
2298 Arg = DAG.getNode(ISD::BITCAST, dl, MVT::getIntegerVT(LocBits), Arg);
2299 Arg = DAG.getNode(ISD::AND, dl, MVT::getIntegerVT(LocBits), Arg, Mask);
2300 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg);
2301 }
2302 }
2303
2304 // f64 and v2f64 might be passed in i32 pairs and must be split into pieces
2305 if (VA.needsCustom() && VA.getLocVT() == MVT::v2f64) {
2306 SDValue Op0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg,
2307 DAG.getConstant(0, dl, MVT::i32));
2308 SDValue Op1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg,
2309 DAG.getConstant(1, dl, MVT::i32));
2310
2311 PassF64ArgInRegs(dl, DAG, Chain, Op0, RegsToPass, VA, ArgLocs[++i],
2312 StackPtr, MemOpChains, isTailCall, SPDiff);
2313
2314 VA = ArgLocs[++i]; // skip ahead to next loc
2315 if (VA.isRegLoc()) {
2316 PassF64ArgInRegs(dl, DAG, Chain, Op1, RegsToPass, VA, ArgLocs[++i],
2317 StackPtr, MemOpChains, isTailCall, SPDiff);
2318 } else {
2319 assert(VA.isMemLoc());
2320 SDValue DstAddr;
2321 MachinePointerInfo DstInfo;
2322 std::tie(DstAddr, DstInfo) =
2323 computeAddrForCallArg(dl, DAG, VA, StackPtr, isTailCall, SPDiff);
2324 MemOpChains.push_back(DAG.getStore(Chain, dl, Op1, DstAddr, DstInfo));
2325 }
2326 } else if (VA.needsCustom() && VA.getLocVT() == MVT::f64) {
2327 PassF64ArgInRegs(dl, DAG, Chain, Arg, RegsToPass, VA, ArgLocs[++i],
2328 StackPtr, MemOpChains, isTailCall, SPDiff);
2329 } else if (VA.isRegLoc()) {
2330 if (realArgIdx == 0 && Flags.isReturned() && !Flags.isSwiftSelf() &&
2331 Outs[0].VT == MVT::i32) {
2332 assert(VA.getLocVT() == MVT::i32 &&
2333 "unexpected calling convention register assignment");
2334 assert(!Ins.empty() && Ins[0].VT == MVT::i32 &&
2335 "unexpected use of 'returned'");
2336 isThisReturn = true;
2337 }
2338 const TargetOptions &Options = DAG.getTarget().Options;
2339 if (Options.EmitCallSiteInfo)
2340 CSInfo.ArgRegPairs.emplace_back(VA.getLocReg(), i);
2341 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg));
2342 } else if (isByVal) {
2343 assert(VA.isMemLoc());
2344 unsigned offset = 0;
2345
2346 // True if this byval aggregate will be split between registers
2347 // and memory.
2348 unsigned ByValArgsCount = CCInfo.getInRegsParamsCount();
2349 unsigned CurByValIdx = CCInfo.getInRegsParamsProcessed();
2350
2351 SDValue ByValSrc;
2352 bool NeedsStackCopy;
2353 if (auto It = ByValTemporaries.find(realArgIdx);
2354 It != ByValTemporaries.end()) {
2355 ByValSrc = It->second;
2356 NeedsStackCopy = true;
2357 } else {
2358 ByValSrc = Arg;
2359 NeedsStackCopy = !isTailCall;
2360 }
2361
2362 // If part of the argument is in registers, load them.
2363 if (CurByValIdx < ByValArgsCount) {
2364 unsigned RegBegin, RegEnd;
2365 CCInfo.getInRegsParamInfo(CurByValIdx, RegBegin, RegEnd);
2366
2367 EVT PtrVT = getPointerTy(DAG.getDataLayout());
2368 unsigned int i, j;
2369 for (i = 0, j = RegBegin; j < RegEnd; i++, j++) {
2370 SDValue Const = DAG.getConstant(4*i, dl, MVT::i32);
2371 SDValue AddArg = DAG.getNode(ISD::ADD, dl, PtrVT, ByValSrc, Const);
2372 SDValue Load =
2373 DAG.getLoad(PtrVT, dl, Chain, AddArg, MachinePointerInfo(),
2374 DAG.InferPtrAlign(AddArg));
2375 MemOpChains.push_back(Load.getValue(1));
2376 RegsToPass.push_back(std::make_pair(j, Load));
2377 }
2378
2379 // If parameter size outsides register area, "offset" value
2380 // helps us to calculate stack slot for remained part properly.
2381 offset = RegEnd - RegBegin;
2382
2383 CCInfo.nextInRegsParam();
2384 }
2385
2386 // If the memory part of the argument isn't already in the correct place
2387 // (which can happen with tail calls), copy it into the argument area.
2388 if (NeedsStackCopy && Flags.getByValSize() > 4 * offset) {
2389 auto PtrVT = getPointerTy(DAG.getDataLayout());
2390 SDValue Dst;
2391 MachinePointerInfo DstInfo;
2392 std::tie(Dst, DstInfo) =
2393 computeAddrForCallArg(dl, DAG, VA, StackPtr, isTailCall, SPDiff);
2394 SDValue SrcOffset = DAG.getIntPtrConstant(4*offset, dl);
2395 SDValue Src = DAG.getNode(ISD::ADD, dl, PtrVT, ByValSrc, SrcOffset);
2396 SDValue SizeNode = DAG.getConstant(Flags.getByValSize() - 4*offset, dl,
2397 MVT::i32);
2398 SDValue AlignNode =
2399 DAG.getConstant(Flags.getNonZeroByValAlign().value(), dl, MVT::i32);
2400
2401 SDVTList VTs = DAG.getVTList(MVT::Other, MVT::Glue);
2402 SDValue Ops[] = { Chain, Dst, Src, SizeNode, AlignNode};
2403 MemOpChains.push_back(DAG.getNode(ARMISD::COPY_STRUCT_BYVAL, dl, VTs,
2404 Ops));
2405 }
2406 } else {
2407 assert(VA.isMemLoc());
2408 SDValue DstAddr;
2409 MachinePointerInfo DstInfo;
2410 std::tie(DstAddr, DstInfo) =
2411 computeAddrForCallArg(dl, DAG, VA, StackPtr, isTailCall, SPDiff);
2412
2413 SDValue Store = DAG.getStore(Chain, dl, Arg, DstAddr, DstInfo);
2414 MemOpChains.push_back(Store);
2415 }
2416 }
2417
2418 if (!MemOpChains.empty())
2419 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOpChains);
2420
2421 // Build a sequence of copy-to-reg nodes chained together with token chain
2422 // and flag operands which copy the outgoing args into the appropriate regs.
2423 SDValue InGlue;
2424 for (const auto &[Reg, N] : RegsToPass) {
2425 Chain = DAG.getCopyToReg(Chain, dl, Reg, N, InGlue);
2426 InGlue = Chain.getValue(1);
2427 }
2428
2429 // If the callee is a GlobalAddress/ExternalSymbol node (quite common, every
2430 // direct call is) turn it into a TargetGlobalAddress/TargetExternalSymbol
2431 // node so that legalize doesn't hack it.
2432 bool isDirect = false;
2433
2434 const TargetMachine &TM = getTargetMachine();
2435 const Triple &TT = TM.getTargetTriple();
2436 const GlobalValue *GVal = nullptr;
2437 if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee))
2438 GVal = G->getGlobal();
2439 bool isStub = !TM.shouldAssumeDSOLocal(GVal) && TT.isOSBinFormatMachO();
2440
2441 bool isARMFunc = !Subtarget->isThumb() || (isStub && !Subtarget->isMClass());
2442 bool isLocalARMFunc = false;
2443 auto PtrVt = getPointerTy(DAG.getDataLayout());
2444
2445 if (Subtarget->genLongCalls()) {
2446 bool isPIC = isPositionIndependent() && !TT.isOSWindows();
2447 if (isPIC && Subtarget->genExecuteOnly())
2448 reportFatalUsageError("long-calls with execute-only and "
2449 "position-independent code is not supported");
2450 if (Subtarget->isROPI())
2451 reportFatalUsageError("long-calls with ROPI is not currently supported");
2452
2453 // Handle a global address or an external symbol. If it's not one of
2454 // those, the target's already in a register, so we don't need to do
2455 // anything extra.
2456 if (isa<GlobalAddressSDNode>(Callee)) {
2457 if (Subtarget->genExecuteOnly()) {
2458 // Execute-only forbids constant pools in .text, so use movw/movt.
2459 // fPIC is not supported with execute-only.
2460 if (Subtarget->useMovt())
2461 ++NumMovwMovt;
2462 Callee = DAG.getNode(ARMISD::Wrapper, dl, PtrVt,
2463 DAG.getTargetGlobalAddress(GVal, dl, PtrVt));
2464 } else if (isPIC) {
2465 // PIC without execute-only: use GOT-based addressing.
2466 // DSO-local symbols use a plain PC-relative WrapperPIC;
2467 // non-DSO-local symbols additionally load the address from the GOT.
2468 SDValue G = DAG.getTargetGlobalAddress(
2469 GVal, dl, PtrVt, 0, GVal->isDSOLocal() ? 0 : ARMII::MO_GOT);
2470 Callee = DAG.getNode(ARMISD::WrapperPIC, dl, PtrVt, G);
2471 if (!GVal->isDSOLocal())
2472 Callee =
2473 DAG.getLoad(PtrVt, dl, DAG.getEntryNode(), Callee,
2475 } else {
2476 // Neither execute-only nor PIC: load the address from a constant pool.
2477 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
2478 ARMConstantPoolValue *CPV = ARMConstantPoolConstant::Create(
2479 GVal, ARMPCLabelIndex, ARMCP::CPValue, 0);
2480
2481 // Get the address of the callee into a register
2482 SDValue Addr = DAG.getTargetConstantPool(CPV, PtrVt, Align(4));
2483 Addr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Addr);
2484 Callee = DAG.getLoad(
2485 PtrVt, dl, DAG.getEntryNode(), Addr,
2487 }
2488 } else if (ExternalSymbolSDNode *S=dyn_cast<ExternalSymbolSDNode>(Callee)) {
2489 const char *Sym = S->getSymbol();
2490
2491 if (Subtarget->genExecuteOnly()) {
2492 // Execute-only forbids constant pools in .text, so use movw/movt.
2493 // fPIC is not supported with execute-only.
2494 if (Subtarget->useMovt())
2495 ++NumMovwMovt;
2496 Callee = DAG.getNode(ARMISD::Wrapper, dl, PtrVt,
2497 DAG.getTargetExternalSymbol(Sym, PtrVt, 0));
2498 } else if (isPIC) {
2499 // PIC without execute-only: load the symbol's address from the GOT via
2500 // a GOT_PREL constant pool entry consumed by a PICLDR.
2501 unsigned PCAdj = Subtarget->isThumb() ? 4 : 8;
2502 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
2503 ARMConstantPoolValue *CPV = ARMConstantPoolSymbol::Create(
2504 *DAG.getContext(), Sym, ARMPCLabelIndex, PCAdj, ARMCP::GOT_PREL,
2505 /*AddCurrentAddress=*/true);
2506 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVt, Align(4));
2507 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
2508 SDValue GOTOffset = DAG.getLoad(
2509 PtrVt, dl, DAG.getEntryNode(), CPAddr,
2511 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32);
2512 Callee = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVt, GOTOffset, PICLabel);
2513 Callee =
2514 DAG.getLoad(PtrVt, dl, DAG.getEntryNode(), Callee,
2516 } else {
2517 // Neither execute-only nor PIC: load the address from a constant pool.
2518 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
2519 ARMConstantPoolValue *CPV = ARMConstantPoolSymbol::Create(
2520 *DAG.getContext(), Sym, ARMPCLabelIndex, 0);
2521
2522 // Get the address of the callee into a register
2523 SDValue Addr = DAG.getTargetConstantPool(CPV, PtrVt, Align(4));
2524 Addr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Addr);
2525 Callee = DAG.getLoad(
2526 PtrVt, dl, DAG.getEntryNode(), Addr,
2528 }
2529 }
2530 } else if (isa<GlobalAddressSDNode>(Callee)) {
2531 if (!PreferIndirect) {
2532 isDirect = true;
2533 bool isDef = GVal->isStrongDefinitionForLinker();
2534
2535 // ARM call to a local ARM function is predicable.
2536 isLocalARMFunc = !Subtarget->isThumb() && (isDef || !ARMInterworking);
2537 // tBX takes a register source operand.
2538 if (isStub && Subtarget->isThumb1Only() && !Subtarget->hasV5TOps()) {
2539 assert(TT.isOSBinFormatMachO() && "WrapperPIC use on non-MachO?");
2540 Callee = DAG.getNode(
2541 ARMISD::WrapperPIC, dl, PtrVt,
2542 DAG.getTargetGlobalAddress(GVal, dl, PtrVt, 0, ARMII::MO_NONLAZY));
2543 Callee = DAG.getLoad(
2544 PtrVt, dl, DAG.getEntryNode(), Callee,
2548 } else if (Subtarget->isTargetCOFF()) {
2549 assert(Subtarget->isTargetWindows() &&
2550 "Windows is the only supported COFF target");
2551 unsigned TargetFlags = ARMII::MO_NO_FLAG;
2552 if (GVal->hasDLLImportStorageClass())
2553 TargetFlags = ARMII::MO_DLLIMPORT;
2554 else if (!TM.shouldAssumeDSOLocal(GVal))
2555 TargetFlags = ARMII::MO_COFFSTUB;
2556 Callee = DAG.getTargetGlobalAddress(GVal, dl, PtrVt, /*offset=*/0,
2557 TargetFlags);
2558 if (TargetFlags & (ARMII::MO_DLLIMPORT | ARMII::MO_COFFSTUB))
2559 Callee =
2560 DAG.getLoad(PtrVt, dl, DAG.getEntryNode(),
2561 DAG.getNode(ARMISD::Wrapper, dl, PtrVt, Callee),
2563 } else {
2564 Callee = DAG.getTargetGlobalAddress(GVal, dl, PtrVt, 0, 0);
2565 }
2566 }
2567 } else if (ExternalSymbolSDNode *S = dyn_cast<ExternalSymbolSDNode>(Callee)) {
2568 isDirect = true;
2569 // tBX takes a register source operand.
2570 const char *Sym = S->getSymbol();
2571 if (isARMFunc && Subtarget->isThumb1Only() && !Subtarget->hasV5TOps()) {
2572 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
2573 ARMConstantPoolValue *CPV =
2575 ARMPCLabelIndex, 4);
2576 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVt, Align(4));
2577 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
2578 Callee = DAG.getLoad(
2579 PtrVt, dl, DAG.getEntryNode(), CPAddr,
2581 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32);
2582 Callee = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVt, Callee, PICLabel);
2583 } else {
2584 Callee = DAG.getTargetExternalSymbol(Sym, PtrVt, 0);
2585 }
2586 }
2587
2588 if (isCmseNSCall) {
2589 assert(!isARMFunc && !isDirect &&
2590 "Cannot handle call to ARM function or direct call");
2591 if (NumBytes > 0) {
2592 DAG.getContext()->diagnose(
2593 DiagnosticInfoUnsupported(DAG.getMachineFunction().getFunction(),
2594 "call to non-secure function would require "
2595 "passing arguments on stack",
2596 dl.getDebugLoc()));
2597 }
2598 if (isStructRet) {
2599 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
2601 "call to non-secure function would return value through pointer",
2602 dl.getDebugLoc()));
2603 }
2604 }
2605
2606 // FIXME: handle tail calls differently.
2607 unsigned CallOpc;
2608 if (Subtarget->isThumb()) {
2609 if (GuardWithBTI)
2610 CallOpc = ARMISD::t2CALL_BTI;
2611 else if (isCmseNSCall)
2612 CallOpc = ARMISD::tSECALL;
2613 else if ((!isDirect || isARMFunc) && !Subtarget->hasV5TOps())
2614 CallOpc = ARMISD::CALL_NOLINK;
2615 else
2616 CallOpc = ARMISD::CALL;
2617 } else {
2618 if (!isDirect && !Subtarget->hasV5TOps())
2619 CallOpc = ARMISD::CALL_NOLINK;
2620 else if (doesNotRet && isDirect && Subtarget->hasRetAddrStack() &&
2621 // Emit regular call when code size is the priority
2622 !Subtarget->hasMinSize())
2623 // "mov lr, pc; b _foo" to avoid confusing the RSP
2624 CallOpc = ARMISD::CALL_NOLINK;
2625 else
2626 CallOpc = isLocalARMFunc ? ARMISD::CALL_PRED : ARMISD::CALL;
2627 }
2628
2629 // We don't usually want to end the call-sequence here because we would tidy
2630 // the frame up *after* the call, however in the ABI-changing tail-call case
2631 // we've carefully laid out the parameters so that when sp is reset they'll be
2632 // in the correct location.
2633 if (isTailCall && !isSibCall) {
2634 Chain = DAG.getCALLSEQ_END(Chain, 0, 0, InGlue, dl);
2635 InGlue = Chain.getValue(1);
2636 }
2637
2638 std::vector<SDValue> Ops;
2639 Ops.push_back(Chain);
2640 Ops.push_back(Callee);
2641
2642 if (isTailCall) {
2643 Ops.push_back(DAG.getSignedTargetConstant(SPDiff, dl, MVT::i32));
2644 }
2645
2646 // Add argument registers to the end of the list so that they are known live
2647 // into the call.
2648 for (const auto &[Reg, N] : RegsToPass)
2649 Ops.push_back(DAG.getRegister(Reg, N.getValueType()));
2650
2651 // Add a register mask operand representing the call-preserved registers.
2652 const uint32_t *Mask;
2653 const ARMBaseRegisterInfo *ARI = Subtarget->getRegisterInfo();
2654 if (isThisReturn) {
2655 // For 'this' returns, use the R0-preserving mask if applicable
2656 Mask = ARI->getThisReturnPreservedMask(MF, CallConv);
2657 if (!Mask) {
2658 // Set isThisReturn to false if the calling convention is not one that
2659 // allows 'returned' to be modeled in this way, so LowerCallResult does
2660 // not try to pass 'this' straight through
2661 isThisReturn = false;
2662 Mask = ARI->getCallPreservedMask(MF, CallConv);
2663 }
2664 } else
2665 Mask = ARI->getCallPreservedMask(MF, CallConv);
2666
2667 assert(Mask && "Missing call preserved mask for calling convention");
2668 Ops.push_back(DAG.getRegisterMask(Mask));
2669
2670 if (InGlue.getNode())
2671 Ops.push_back(InGlue);
2672
2673 if (isTailCall) {
2675 SDValue Ret = DAG.getNode(ARMISD::TC_RETURN, dl, MVT::Other, Ops);
2676 if (CLI.CFIType)
2677 Ret.getNode()->setCFIType(CLI.CFIType->getZExtValue());
2678 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
2679 DAG.addCallSiteInfo(Ret.getNode(), std::move(CSInfo));
2680 return Ret;
2681 }
2682
2683 // Returns a chain and a flag for retval copy to use.
2684 Chain = DAG.getNode(CallOpc, dl, {MVT::Other, MVT::Glue}, Ops);
2685 if (CLI.CFIType)
2686 Chain.getNode()->setCFIType(CLI.CFIType->getZExtValue());
2687 DAG.addNoMergeSiteInfo(Chain.getNode(), CLI.NoMerge);
2688 InGlue = Chain.getValue(1);
2689 DAG.addCallSiteInfo(Chain.getNode(), std::move(CSInfo));
2690
2691 // If we're guaranteeing tail-calls will be honoured, the callee must
2692 // pop its own argument stack on return. But this call is *not* a tail call so
2693 // we need to undo that after it returns to restore the status-quo.
2694 bool TailCallOpt = getTargetMachine().Options.GuaranteedTailCallOpt;
2695 uint64_t CalleePopBytes =
2696 canGuaranteeTCO(CallConv, TailCallOpt) ? alignTo(NumBytes, 16) : -1U;
2697
2698 Chain = DAG.getCALLSEQ_END(Chain, NumBytes, CalleePopBytes, InGlue, dl);
2699 if (!Ins.empty())
2700 InGlue = Chain.getValue(1);
2701
2702 // Handle result values, copying them out of physregs into vregs that we
2703 // return.
2704 return LowerCallResult(Chain, InGlue, CallConv, isVarArg, Ins, dl, DAG,
2705 InVals, isThisReturn,
2706 isThisReturn ? OutVals[0] : SDValue(), isCmseNSCall);
2707}
2708
2709/// HandleByVal - Every parameter *after* a byval parameter is passed
2710/// on the stack. Remember the next parameter register to allocate,
2711/// and then confiscate the rest of the parameter registers to insure
2712/// this.
2713void ARMTargetLowering::HandleByVal(CCState *State, unsigned &Size,
2714 Align Alignment) const {
2715 // Byval (as with any stack) slots are always at least 4 byte aligned.
2716 Alignment = std::max(Alignment, Align(4));
2717
2718 MCRegister Reg = State->AllocateReg(GPRArgRegs);
2719 if (!Reg)
2720 return;
2721
2722 unsigned AlignInRegs = Alignment.value() / 4;
2723 unsigned Waste = (ARM::R4 - Reg) % AlignInRegs;
2724 for (unsigned i = 0; i < Waste; ++i)
2725 Reg = State->AllocateReg(GPRArgRegs);
2726
2727 if (!Reg)
2728 return;
2729
2730 unsigned Excess = 4 * (ARM::R4 - Reg);
2731
2732 // Special case when NSAA != SP and parameter size greater than size of
2733 // all remained GPR regs. In that case we can't split parameter, we must
2734 // send it to stack. We also must set NCRN to R4, so waste all
2735 // remained registers.
2736 const unsigned NSAAOffset = State->getStackSize();
2737 if (NSAAOffset != 0 && Size > Excess) {
2738 while (State->AllocateReg(GPRArgRegs))
2739 ;
2740 return;
2741 }
2742
2743 // First register for byval parameter is the first register that wasn't
2744 // allocated before this method call, so it would be "reg".
2745 // If parameter is small enough to be saved in range [reg, r4), then
2746 // the end (first after last) register would be reg + param-size-in-regs,
2747 // else parameter would be splitted between registers and stack,
2748 // end register would be r4 in this case.
2749 unsigned ByValRegBegin = Reg;
2750 unsigned ByValRegEnd = std::min<unsigned>(Reg + Size / 4, ARM::R4);
2751 State->addInRegsParamInfo(ByValRegBegin, ByValRegEnd);
2752 // Note, first register is allocated in the beginning of function already,
2753 // allocate remained amount of registers we need.
2754 for (unsigned i = Reg + 1; i != ByValRegEnd; ++i)
2755 State->AllocateReg(GPRArgRegs);
2756 // A byval parameter that is split between registers and memory needs its
2757 // size truncated here.
2758 // In the case where the entire structure fits in registers, we set the
2759 // size in memory to zero.
2760 Size = std::max<int>(Size - Excess, 0);
2761}
2762
2763/// IsEligibleForTailCallOptimization - Check whether the call is eligible
2764/// for tail call optimization. Targets which want to do tail call
2765/// optimization should implement this function. Note that this function also
2766/// processes musttail calls, so when this function returns false on a valid
2767/// musttail call, a fatal backend error occurs.
2768bool ARMTargetLowering::IsEligibleForTailCallOptimization(
2770 SmallVectorImpl<CCValAssign> &ArgLocs, const bool isIndirect) const {
2771 CallingConv::ID CalleeCC = CLI.CallConv;
2772 SDValue Callee = CLI.Callee;
2773 bool isVarArg = CLI.IsVarArg;
2774 const SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
2775 const SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2776 const SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
2777 const SelectionDAG &DAG = CLI.DAG;
2779 const Function &CallerF = MF.getFunction();
2780 CallingConv::ID CallerCC = CallerF.getCallingConv();
2781
2782 assert(Subtarget->supportsTailCall());
2783
2784 // Indirect tail-calls require a register to hold the target address. That
2785 // register must be:
2786 // * Allocatable (i.e. r0-r7 if the target is Thumb1).
2787 // * Not callee-saved, so must be one of r0-r3 or r12.
2788 // * Not used to hold an argument to the tail-called function, which might be
2789 // in r0-r3.
2790 // * Not used to hold the return address authentication code, which is in r12
2791 // if enabled.
2792 // Sometimes, no register matches all of these conditions, so we can't do a
2793 // tail-call.
2794 if (!isa<GlobalAddressSDNode>(Callee.getNode()) || isIndirect) {
2795 SmallSet<MCPhysReg, 5> AddressRegisters = {ARM::R0, ARM::R1, ARM::R2,
2796 ARM::R3};
2797 if (!(Subtarget->isThumb1Only() ||
2798 MF.getInfo<ARMFunctionInfo>()->shouldSignReturnAddress(true)))
2799 AddressRegisters.insert(ARM::R12);
2800 for (const CCValAssign &AL : ArgLocs)
2801 if (AL.isRegLoc())
2802 AddressRegisters.erase(AL.getLocReg());
2803 if (AddressRegisters.empty()) {
2804 LLVM_DEBUG(dbgs() << "false (no reg to hold function pointer)\n");
2805 return false;
2806 }
2807 }
2808
2809 // Look for obvious safe cases to perform tail call optimization that do not
2810 // require ABI changes. This is what gcc calls sibcall.
2811
2812 // Exception-handling functions need a special set of instructions to indicate
2813 // a return to the hardware. Tail-calling another function would probably
2814 // break this.
2815 if (CallerF.hasFnAttribute("interrupt")) {
2816 LLVM_DEBUG(dbgs() << "false (interrupt attribute)\n");
2817 return false;
2818 }
2819
2820 if (canGuaranteeTCO(CalleeCC,
2821 getTargetMachine().Options.GuaranteedTailCallOpt)) {
2822 LLVM_DEBUG(dbgs() << (CalleeCC == CallerCC ? "true" : "false")
2823 << " (guaranteed tail-call CC)\n");
2824 return CalleeCC == CallerCC;
2825 }
2826
2827 // Also avoid sibcall optimization if either caller or callee uses struct
2828 // return semantics.
2829 bool isCalleeStructRet = Outs.empty() ? false : Outs[0].Flags.isSRet();
2830 bool isCallerStructRet = MF.getFunction().hasStructRetAttr();
2831 if (isCalleeStructRet != isCallerStructRet) {
2832 LLVM_DEBUG(dbgs() << "false (struct-ret)\n");
2833 return false;
2834 }
2835
2836 // Externally-defined functions with weak linkage should not be
2837 // tail-called on ARM when the OS does not support dynamic
2838 // pre-emption of symbols, as the AAELF spec requires normal calls
2839 // to undefined weak functions to be replaced with a NOP or jump to the
2840 // next instruction. The behaviour of branch instructions in this
2841 // situation (as used for tail calls) is implementation-defined, so we
2842 // cannot rely on the linker replacing the tail call with a return.
2843 if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
2844 const GlobalValue *GV = G->getGlobal();
2845 const Triple &TT = GV->getParent()->getTargetTriple();
2846 if (GV->hasExternalWeakLinkage() &&
2847 (!TT.isOSWindows() || TT.isOSBinFormatELF() ||
2848 TT.isOSBinFormatMachO())) {
2849 LLVM_DEBUG(dbgs() << "false (external weak linkage)\n");
2850 return false;
2851 }
2852 }
2853
2854 // Check that the call results are passed in the same way.
2855 LLVMContext &C = *DAG.getContext();
2857 getEffectiveCallingConv(CalleeCC, isVarArg),
2858 getEffectiveCallingConv(CallerCC, CallerF.isVarArg()), MF, C, Ins,
2859 CCAssignFnForReturn(CalleeCC, isVarArg),
2860 CCAssignFnForReturn(CallerCC, CallerF.isVarArg()))) {
2861 LLVM_DEBUG(dbgs() << "false (incompatible results)\n");
2862 return false;
2863 }
2864 // The callee has to preserve all registers the caller needs to preserve.
2865 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo();
2866 const uint32_t *CallerPreserved = TRI->getCallPreservedMask(MF, CallerCC);
2867 if (CalleeCC != CallerCC) {
2868 const uint32_t *CalleePreserved = TRI->getCallPreservedMask(MF, CalleeCC);
2869 if (!TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved)) {
2870 LLVM_DEBUG(dbgs() << "false (not all registers preserved)\n");
2871 return false;
2872 }
2873 }
2874
2875 // If Caller's vararg argument has been split between registers and stack, do
2876 // not perform tail call, since part of the argument is in caller's local
2877 // frame.
2878 const ARMFunctionInfo *AFI_Caller = MF.getInfo<ARMFunctionInfo>();
2879 if (CLI.IsVarArg && AFI_Caller->getArgRegsSaveSize()) {
2880 LLVM_DEBUG(dbgs() << "false (arg reg save area)\n");
2881 return false;
2882 }
2883
2884 // If the callee takes no arguments then go on to check the results of the
2885 // call.
2886 const MachineRegisterInfo &MRI = MF.getRegInfo();
2887 if (!parametersInCSRMatch(MRI, CallerPreserved, ArgLocs, OutVals)) {
2888 LLVM_DEBUG(dbgs() << "false (parameters in CSRs do not match)\n");
2889 return false;
2890 }
2891
2892 // If the stack arguments for this call do not fit into our own save area then
2893 // the call cannot be made tail.
2894 if (CCInfo.getStackSize() > AFI_Caller->getArgumentStackSize())
2895 return false;
2896
2897 LLVM_DEBUG(dbgs() << "true\n");
2898 return true;
2899}
2900
2901bool
2902ARMTargetLowering::CanLowerReturn(CallingConv::ID CallConv,
2903 MachineFunction &MF, bool isVarArg,
2905 LLVMContext &Context, const Type *RetTy) const {
2907 CCState CCInfo(CallConv, isVarArg, MF, RVLocs, Context);
2908 return CCInfo.CheckReturn(Outs, CCAssignFnForReturn(CallConv, isVarArg));
2909}
2910
2912 const SDLoc &DL, SelectionDAG &DAG) {
2913 const MachineFunction &MF = DAG.getMachineFunction();
2914 const Function &F = MF.getFunction();
2915
2916 StringRef IntKind = F.getFnAttribute("interrupt").getValueAsString();
2917
2918 // See ARM ARM v7 B1.8.3. On exception entry LR is set to a possibly offset
2919 // version of the "preferred return address". These offsets affect the return
2920 // instruction if this is a return from PL1 without hypervisor extensions.
2921 // IRQ/FIQ: +4 "subs pc, lr, #4"
2922 // SWI: 0 "subs pc, lr, #0"
2923 // ABORT: +4 "subs pc, lr, #4"
2924 // UNDEF: +4/+2 "subs pc, lr, #0"
2925 // UNDEF varies depending on where the exception came from ARM or Thumb
2926 // mode. Alongside GCC, we throw our hands up in disgust and pretend it's 0.
2927
2928 int64_t LROffset;
2929 if (IntKind == "" || IntKind == "IRQ" || IntKind == "FIQ" ||
2930 IntKind == "ABORT")
2931 LROffset = 4;
2932 else if (IntKind == "SWI" || IntKind == "UNDEF")
2933 LROffset = 0;
2934 else
2935 report_fatal_error("Unsupported interrupt attribute. If present, value "
2936 "must be one of: IRQ, FIQ, SWI, ABORT or UNDEF");
2937
2938 RetOps.insert(RetOps.begin() + 1,
2939 DAG.getConstant(LROffset, DL, MVT::i32, false));
2940
2941 return DAG.getNode(ARMISD::INTRET_GLUE, DL, MVT::Other, RetOps);
2942}
2943
2944SDValue
2945ARMTargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv,
2946 bool isVarArg,
2948 const SmallVectorImpl<SDValue> &OutVals,
2949 const SDLoc &dl, SelectionDAG &DAG) const {
2950 // CCValAssign - represent the assignment of the return value to a location.
2952
2953 // CCState - Info about the registers and stack slots.
2954 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs,
2955 *DAG.getContext());
2956
2957 // Analyze outgoing return values.
2958 CCInfo.AnalyzeReturn(Outs, CCAssignFnForReturn(CallConv, isVarArg));
2959
2960 SDValue Glue;
2962 RetOps.push_back(Chain); // Operand #0 = Chain (updated below)
2963 bool isLittleEndian = Subtarget->isLittle();
2964
2966 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
2967 AFI->setReturnRegsCount(RVLocs.size());
2968
2969 // Report error if cmse entry function returns structure through first ptr arg.
2970 if (AFI->isCmseNSEntryFunction() && MF.getFunction().hasStructRetAttr()) {
2971 // Note: using an empty SDLoc(), as the first line of the function is a
2972 // better place to report than the last line.
2973 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
2975 "secure entry function would return value through pointer",
2976 SDLoc().getDebugLoc()));
2977 }
2978
2979 // Copy the result values into the output registers.
2980 for (unsigned i = 0, realRVLocIdx = 0;
2981 i != RVLocs.size();
2982 ++i, ++realRVLocIdx) {
2983 CCValAssign &VA = RVLocs[i];
2984 assert(VA.isRegLoc() && "Can only return in registers!");
2985
2986 SDValue Arg = OutVals[realRVLocIdx];
2987 bool ReturnF16 = false;
2988
2989 if (Subtarget->hasFullFP16() && Subtarget->isTargetHardFloat()) {
2990 // Half-precision return values can be returned like this:
2991 //
2992 // t11 f16 = fadd ...
2993 // t12: i16 = bitcast t11
2994 // t13: i32 = zero_extend t12
2995 // t14: f32 = bitcast t13 <~~~~~~~ Arg
2996 //
2997 // to avoid code generation for bitcasts, we simply set Arg to the node
2998 // that produces the f16 value, t11 in this case.
2999 //
3000 if (Arg.getValueType() == MVT::f32 && Arg.getOpcode() == ISD::BITCAST) {
3001 SDValue ZE = Arg.getOperand(0);
3002 if (ZE.getOpcode() == ISD::ZERO_EXTEND && ZE.getValueType() == MVT::i32) {
3003 SDValue BC = ZE.getOperand(0);
3004 if (BC.getOpcode() == ISD::BITCAST && BC.getValueType() == MVT::i16) {
3005 Arg = BC.getOperand(0);
3006 ReturnF16 = true;
3007 }
3008 }
3009 }
3010 }
3011
3012 switch (VA.getLocInfo()) {
3013 default: llvm_unreachable("Unknown loc info!");
3014 case CCValAssign::Full: break;
3015 case CCValAssign::BCvt:
3016 if (!ReturnF16)
3017 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg);
3018 break;
3019 }
3020
3021 // Mask f16 arguments if this is a CMSE nonsecure entry.
3022 auto RetVT = Outs[realRVLocIdx].ArgVT;
3023 if (AFI->isCmseNSEntryFunction() && (RetVT == MVT::f16)) {
3024 if (VA.needsCustom() && VA.getValVT() == MVT::f16) {
3025 Arg = MoveFromHPR(dl, DAG, VA.getLocVT(), VA.getValVT(), Arg);
3026 } else {
3027 auto LocBits = VA.getLocVT().getSizeInBits();
3028 auto MaskValue = APInt::getLowBitsSet(LocBits, RetVT.getSizeInBits());
3029 SDValue Mask =
3030 DAG.getConstant(MaskValue, dl, MVT::getIntegerVT(LocBits));
3031 Arg = DAG.getNode(ISD::BITCAST, dl, MVT::getIntegerVT(LocBits), Arg);
3032 Arg = DAG.getNode(ISD::AND, dl, MVT::getIntegerVT(LocBits), Arg, Mask);
3033 Arg = DAG.getNode(ISD::BITCAST, dl, VA.getLocVT(), Arg);
3034 }
3035 }
3036
3037 if (VA.needsCustom() &&
3038 (VA.getLocVT() == MVT::v2f64 || VA.getLocVT() == MVT::f64)) {
3039 if (VA.getLocVT() == MVT::v2f64) {
3040 // Extract the first half and return it in two registers.
3041 SDValue Half = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg,
3042 DAG.getConstant(0, dl, MVT::i32));
3043 SDValue HalfGPRs = DAG.getNode(ARMISD::VMOVRRD, dl,
3044 DAG.getVTList(MVT::i32, MVT::i32), Half);
3045
3046 Chain =
3047 DAG.getCopyToReg(Chain, dl, VA.getLocReg(),
3048 HalfGPRs.getValue(isLittleEndian ? 0 : 1), Glue);
3049 Glue = Chain.getValue(1);
3050 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
3051 VA = RVLocs[++i]; // skip ahead to next loc
3052 Chain =
3053 DAG.getCopyToReg(Chain, dl, VA.getLocReg(),
3054 HalfGPRs.getValue(isLittleEndian ? 1 : 0), Glue);
3055 Glue = Chain.getValue(1);
3056 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
3057 VA = RVLocs[++i]; // skip ahead to next loc
3058
3059 // Extract the 2nd half and fall through to handle it as an f64 value.
3060 Arg = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Arg,
3061 DAG.getConstant(1, dl, MVT::i32));
3062 }
3063 // Legalize ret f64 -> ret 2 x i32. We always have fmrrd if f64 is
3064 // available.
3065 SDValue fmrrd = DAG.getNode(ARMISD::VMOVRRD, dl,
3066 DAG.getVTList(MVT::i32, MVT::i32), Arg);
3067 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(),
3068 fmrrd.getValue(isLittleEndian ? 0 : 1), Glue);
3069 Glue = Chain.getValue(1);
3070 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
3071 VA = RVLocs[++i]; // skip ahead to next loc
3072 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(),
3073 fmrrd.getValue(isLittleEndian ? 1 : 0), Glue);
3074 } else
3075 Chain = DAG.getCopyToReg(Chain, dl, VA.getLocReg(), Arg, Glue);
3076
3077 // Guarantee that all emitted copies are
3078 // stuck together, avoiding something bad.
3079 Glue = Chain.getValue(1);
3080 RetOps.push_back(DAG.getRegister(
3081 VA.getLocReg(), ReturnF16 ? Arg.getValueType() : VA.getLocVT()));
3082 }
3083 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo();
3084 const MCPhysReg *I =
3085 TRI->getCalleeSavedRegsViaCopy(&DAG.getMachineFunction());
3086 if (I) {
3087 for (; *I; ++I) {
3088 if (ARM::GPRRegClass.contains(*I))
3089 RetOps.push_back(DAG.getRegister(*I, MVT::i32));
3090 else if (ARM::DPRRegClass.contains(*I))
3092 else
3093 llvm_unreachable("Unexpected register class in CSRsViaCopy!");
3094 }
3095 }
3096
3097 // Update chain and glue.
3098 RetOps[0] = Chain;
3099 if (Glue.getNode())
3100 RetOps.push_back(Glue);
3101
3102 // CPUs which aren't M-class use a special sequence to return from
3103 // exceptions (roughly, any instruction setting pc and cpsr simultaneously,
3104 // though we use "subs pc, lr, #N").
3105 //
3106 // M-class CPUs actually use a normal return sequence with a special
3107 // (hardware-provided) value in LR, so the normal code path works.
3108 if (DAG.getMachineFunction().getFunction().hasFnAttribute("interrupt") &&
3109 !Subtarget->isMClass()) {
3110 if (Subtarget->isThumb1Only())
3111 report_fatal_error("interrupt attribute is not supported in Thumb1");
3112 return LowerInterruptReturn(RetOps, dl, DAG);
3113 }
3114
3115 unsigned RetNode =
3116 AFI->isCmseNSEntryFunction() ? ARMISD::SERET_GLUE : ARMISD::RET_GLUE;
3117 return DAG.getNode(RetNode, dl, MVT::Other, RetOps);
3118}
3119
3120bool ARMTargetLowering::isUsedByReturnOnly(SDNode *N, SDValue &Chain) const {
3121 if (N->getNumValues() != 1)
3122 return false;
3123 if (!N->hasNUsesOfValue(1, 0))
3124 return false;
3125
3126 SDValue TCChain = Chain;
3127 SDNode *Copy = *N->user_begin();
3128 if (Copy->getOpcode() == ISD::CopyToReg) {
3129 // If the copy has a glue operand, we conservatively assume it isn't safe to
3130 // perform a tail call.
3131 if (Copy->getOperand(Copy->getNumOperands()-1).getValueType() == MVT::Glue)
3132 return false;
3133 TCChain = Copy->getOperand(0);
3134 } else if (Copy->getOpcode() == ARMISD::VMOVRRD) {
3135 SDNode *VMov = Copy;
3136 // f64 returned in a pair of GPRs.
3137 SmallPtrSet<SDNode*, 2> Copies;
3138 for (SDNode *U : VMov->users()) {
3139 if (U->getOpcode() != ISD::CopyToReg)
3140 return false;
3141 Copies.insert(U);
3142 }
3143 if (Copies.size() > 2)
3144 return false;
3145
3146 for (SDNode *U : VMov->users()) {
3147 SDValue UseChain = U->getOperand(0);
3148 if (Copies.count(UseChain.getNode()))
3149 // Second CopyToReg
3150 Copy = U;
3151 else {
3152 // We are at the top of this chain.
3153 // If the copy has a glue operand, we conservatively assume it
3154 // isn't safe to perform a tail call.
3155 if (U->getOperand(U->getNumOperands() - 1).getValueType() == MVT::Glue)
3156 return false;
3157 // First CopyToReg
3158 TCChain = UseChain;
3159 }
3160 }
3161 } else if (Copy->getOpcode() == ISD::BITCAST) {
3162 // f32 returned in a single GPR.
3163 if (!Copy->hasOneUse())
3164 return false;
3165 Copy = *Copy->user_begin();
3166 if (Copy->getOpcode() != ISD::CopyToReg || !Copy->hasNUsesOfValue(1, 0))
3167 return false;
3168 // If the copy has a glue operand, we conservatively assume it isn't safe to
3169 // perform a tail call.
3170 if (Copy->getOperand(Copy->getNumOperands()-1).getValueType() == MVT::Glue)
3171 return false;
3172 TCChain = Copy->getOperand(0);
3173 } else {
3174 return false;
3175 }
3176
3177 bool HasRet = false;
3178 for (const SDNode *U : Copy->users()) {
3179 if (U->getOpcode() != ARMISD::RET_GLUE &&
3180 U->getOpcode() != ARMISD::INTRET_GLUE)
3181 return false;
3182 HasRet = true;
3183 }
3184
3185 if (!HasRet)
3186 return false;
3187
3188 Chain = TCChain;
3189 return true;
3190}
3191
3192bool ARMTargetLowering::mayBeEmittedAsTailCall(const CallInst *CI) const {
3193 if (!Subtarget->supportsTailCall())
3194 return false;
3195
3196 if (!CI->isTailCall())
3197 return false;
3198
3199 return true;
3200}
3201
3202// Trying to write a 64 bit value so need to split into two 32 bit values first,
3203// and pass the lower and high parts through.
3205 SDLoc DL(Op);
3206 SDValue WriteValue = Op->getOperand(2);
3207
3208 // This function is only supposed to be called for i64 type argument.
3209 assert(WriteValue.getValueType() == MVT::i64
3210 && "LowerWRITE_REGISTER called for non-i64 type argument.");
3211
3212 SDValue Lo, Hi;
3213 std::tie(Lo, Hi) = DAG.SplitScalar(WriteValue, DL, MVT::i32, MVT::i32);
3214 SDValue Ops[] = { Op->getOperand(0), Op->getOperand(1), Lo, Hi };
3215 return DAG.getNode(ISD::WRITE_REGISTER, DL, MVT::Other, Ops);
3216}
3217
3218// ConstantPool, JumpTable, GlobalAddress, and ExternalSymbol are lowered as
3219// their target counterpart wrapped in the ARMISD::Wrapper node. Suppose N is
3220// one of the above mentioned nodes. It has to be wrapped because otherwise
3221// Select(N) returns N. So the raw TargetGlobalAddress nodes, etc. can only
3222// be used to form addressing mode. These wrapped nodes will be selected
3223// into MOVi.
3224SDValue ARMTargetLowering::LowerConstantPool(SDValue Op,
3225 SelectionDAG &DAG) const {
3226 EVT PtrVT = Op.getValueType();
3227 // FIXME there is no actual debug info here
3228 SDLoc dl(Op);
3229 ConstantPoolSDNode *CP = cast<ConstantPoolSDNode>(Op);
3230 SDValue Res;
3231
3232 // When generating execute-only code Constant Pools must be promoted to the
3233 // global data section. It's a bit ugly that we can't share them across basic
3234 // blocks, but this way we guarantee that execute-only behaves correct with
3235 // position-independent addressing modes.
3236 if (Subtarget->genExecuteOnly()) {
3237 auto AFI = DAG.getMachineFunction().getInfo<ARMFunctionInfo>();
3238 auto *T = CP->getType();
3239 auto C = const_cast<Constant*>(CP->getConstVal());
3240 auto M = DAG.getMachineFunction().getFunction().getParent();
3241 auto GV = new GlobalVariable(
3242 *M, T, /*isConstant=*/true, GlobalVariable::InternalLinkage, C,
3243 Twine(DAG.getDataLayout().getInternalSymbolPrefix()) + "CP" +
3244 Twine(DAG.getMachineFunction().getFunctionNumber()) + "_" +
3245 Twine(AFI->createPICLabelUId()));
3246 SDValue GA = DAG.getTargetGlobalAddress(GV, dl, PtrVT);
3247 return LowerGlobalAddress(GA, DAG);
3248 }
3249
3250 // The 16-bit ADR instruction can only encode offsets that are multiples of 4,
3251 // so we need to align to at least 4 bytes when we don't have 32-bit ADR.
3252 Align CPAlign = CP->getAlign();
3253 if (Subtarget->isThumb1Only())
3254 CPAlign = std::max(CPAlign, Align(4));
3256 Res =
3257 DAG.getTargetConstantPool(CP->getMachineCPVal(), PtrVT, CPAlign);
3258 else
3259 Res = DAG.getTargetConstantPool(CP->getConstVal(), PtrVT, CPAlign);
3260 return DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Res);
3261}
3262
3264 // If we don't have a 32-bit pc-relative branch instruction then the jump
3265 // table consists of block addresses. Usually this is inline, but for
3266 // execute-only it must be placed out-of-line.
3267 if (Subtarget->genExecuteOnly() && !Subtarget->hasV8MBaselineOps())
3270}
3271
3272SDValue ARMTargetLowering::LowerBlockAddress(SDValue Op,
3273 SelectionDAG &DAG) const {
3276 unsigned ARMPCLabelIndex = 0;
3277 SDLoc DL(Op);
3278 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3279 const BlockAddress *BA = cast<BlockAddressSDNode>(Op)->getBlockAddress();
3280 SDValue CPAddr;
3281 bool IsPositionIndependent = isPositionIndependent() || Subtarget->isROPI();
3282 if (!IsPositionIndependent) {
3283 CPAddr = DAG.getTargetConstantPool(BA, PtrVT, Align(4));
3284 } else {
3285 unsigned PCAdj = Subtarget->isThumb() ? 4 : 8;
3286 ARMPCLabelIndex = AFI->createPICLabelUId();
3288 ARMConstantPoolConstant::Create(BA, ARMPCLabelIndex,
3289 ARMCP::CPBlockAddress, PCAdj);
3290 CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3291 }
3292 CPAddr = DAG.getNode(ARMISD::Wrapper, DL, PtrVT, CPAddr);
3293 SDValue Result = DAG.getLoad(
3294 PtrVT, DL, DAG.getEntryNode(), CPAddr,
3296 if (!IsPositionIndependent)
3297 return Result;
3298 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, DL, MVT::i32);
3299 return DAG.getNode(ARMISD::PIC_ADD, DL, PtrVT, Result, PICLabel);
3300}
3301
3302/// Convert a TLS address reference into the correct sequence of loads
3303/// and calls to compute the variable's address for Darwin, and return an
3304/// SDValue containing the final node.
3305
3306/// Darwin only has one TLS scheme which must be capable of dealing with the
3307/// fully general situation, in the worst case. This means:
3308/// + "extern __thread" declaration.
3309/// + Defined in a possibly unknown dynamic library.
3310///
3311/// The general system is that each __thread variable has a [3 x i32] descriptor
3312/// which contains information used by the runtime to calculate the address. The
3313/// only part of this the compiler needs to know about is the first word, which
3314/// contains a function pointer that must be called with the address of the
3315/// entire descriptor in "r0".
3316///
3317/// Since this descriptor may be in a different unit, in general access must
3318/// proceed along the usual ARM rules. A common sequence to produce is:
3319///
3320/// movw rT1, :lower16:_var$non_lazy_ptr
3321/// movt rT1, :upper16:_var$non_lazy_ptr
3322/// ldr r0, [rT1]
3323/// ldr rT2, [r0]
3324/// blx rT2
3325/// [...address now in r0...]
3326SDValue
3327ARMTargetLowering::LowerGlobalTLSAddressDarwin(SDValue Op,
3328 SelectionDAG &DAG) const {
3329 assert(getTargetMachine().getTargetTriple().isOSDarwin() &&
3330 "This function expects a Darwin target");
3331 SDLoc DL(Op);
3332
3333 // First step is to get the address of the actua global symbol. This is where
3334 // the TLS descriptor lives.
3335 SDValue DescAddr = LowerGlobalAddressDarwin(Op, DAG);
3336
3337 // The first entry in the descriptor is a function pointer that we must call
3338 // to obtain the address of the variable.
3339 SDValue Chain = DAG.getEntryNode();
3340 SDValue FuncTLVGet = DAG.getLoad(
3341 MVT::i32, DL, Chain, DescAddr,
3345 Chain = FuncTLVGet.getValue(1);
3346
3348 MachineFrameInfo &MFI = F.getFrameInfo();
3349 MFI.setAdjustsStack(true);
3350
3351 // TLS calls preserve all registers except those that absolutely must be
3352 // trashed: R0 (it takes an argument), LR (it's a call) and CPSR (let's not be
3353 // silly).
3354 auto TRI =
3356 auto ARI = static_cast<const ARMRegisterInfo *>(TRI);
3357 const uint32_t *Mask = ARI->getTLSCallPreservedMask(DAG.getMachineFunction());
3358
3359 // Finally, we can make the call. This is just a degenerate version of a
3360 // normal AArch64 call node: r0 takes the address of the descriptor, and
3361 // returns the address of the variable in this thread.
3362 Chain = DAG.getCopyToReg(Chain, DL, ARM::R0, DescAddr, SDValue());
3363 Chain =
3364 DAG.getNode(ARMISD::CALL, DL, DAG.getVTList(MVT::Other, MVT::Glue),
3365 Chain, FuncTLVGet, DAG.getRegister(ARM::R0, MVT::i32),
3366 DAG.getRegisterMask(Mask), Chain.getValue(1));
3367 return DAG.getCopyFromReg(Chain, DL, ARM::R0, MVT::i32, Chain.getValue(1));
3368}
3369
3370SDValue
3371ARMTargetLowering::LowerGlobalTLSAddressWindows(SDValue Op,
3372 SelectionDAG &DAG) const {
3373 assert(getTargetMachine().getTargetTriple().isOSWindows() &&
3374 "Windows specific TLS lowering");
3375
3376 SDValue Chain = DAG.getEntryNode();
3377 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3378 SDLoc DL(Op);
3379
3380 // Load the current TEB (thread environment block)
3381 SDValue Ops[] = {Chain,
3382 DAG.getTargetConstant(Intrinsic::arm_mrc, DL, MVT::i32),
3383 DAG.getTargetConstant(15, DL, MVT::i32),
3384 DAG.getTargetConstant(0, DL, MVT::i32),
3385 DAG.getTargetConstant(13, DL, MVT::i32),
3386 DAG.getTargetConstant(0, DL, MVT::i32),
3387 DAG.getTargetConstant(2, DL, MVT::i32)};
3388 SDValue CurrentTEB = DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL,
3389 DAG.getVTList(MVT::i32, MVT::Other), Ops);
3390
3391 SDValue TEB = CurrentTEB.getValue(0);
3392 Chain = CurrentTEB.getValue(1);
3393
3394 // Load the ThreadLocalStoragePointer from the TEB
3395 // A pointer to the TLS array is located at offset 0x2c from the TEB.
3396 SDValue TLSArray =
3397 DAG.getNode(ISD::ADD, DL, PtrVT, TEB, DAG.getIntPtrConstant(0x2c, DL));
3398 TLSArray = DAG.getLoad(PtrVT, DL, Chain, TLSArray, MachinePointerInfo());
3399
3400 // The pointer to the thread's TLS data area is at the TLS Index scaled by 4
3401 // offset into the TLSArray.
3402
3403 // Load the TLS index from the C runtime
3404 SDValue TLSIndex =
3405 DAG.getTargetExternalSymbol("_tls_index", PtrVT, ARMII::MO_NO_FLAG);
3406 TLSIndex = DAG.getNode(ARMISD::Wrapper, DL, PtrVT, TLSIndex);
3407 TLSIndex = DAG.getLoad(PtrVT, DL, Chain, TLSIndex, MachinePointerInfo());
3408
3409 SDValue Slot = DAG.getNode(ISD::SHL, DL, PtrVT, TLSIndex,
3410 DAG.getConstant(2, DL, MVT::i32));
3411 SDValue TLS = DAG.getLoad(PtrVT, DL, Chain,
3412 DAG.getNode(ISD::ADD, DL, PtrVT, TLSArray, Slot),
3413 MachinePointerInfo());
3414
3415 // Get the offset of the start of the .tls section (section base)
3416 const auto *GA = cast<GlobalAddressSDNode>(Op);
3417 auto *CPV = ARMConstantPoolConstant::Create(GA->getGlobal(), ARMCP::SECREL);
3418 SDValue Offset = DAG.getLoad(
3419 PtrVT, DL, Chain,
3420 DAG.getNode(ARMISD::Wrapper, DL, MVT::i32,
3421 DAG.getTargetConstantPool(CPV, PtrVT, Align(4))),
3423
3424 return DAG.getNode(ISD::ADD, DL, PtrVT, TLS, Offset);
3425}
3426
3427// Lower ISD::GlobalTLSAddress using the "general dynamic" model
3428SDValue
3429ARMTargetLowering::LowerToTLSGeneralDynamicModel(GlobalAddressSDNode *GA,
3430 SelectionDAG &DAG) const {
3431 SDLoc dl(GA);
3432 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3433 unsigned char PCAdj = Subtarget->isThumb() ? 4 : 8;
3435 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
3436 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
3437 ARMConstantPoolValue *CPV =
3438 ARMConstantPoolConstant::Create(GA->getGlobal(), ARMPCLabelIndex,
3439 ARMCP::CPValue, PCAdj, ARMCP::TLSGD, true);
3440 SDValue Argument = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3441 Argument = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Argument);
3442 Argument = DAG.getLoad(
3443 PtrVT, dl, DAG.getEntryNode(), Argument,
3445 SDValue Chain = Argument.getValue(1);
3446
3447 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32);
3448 Argument = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Argument, PICLabel);
3449
3450 // call __tls_get_addr.
3452 Args.emplace_back(Argument, Type::getInt32Ty(*DAG.getContext()));
3453
3454 // FIXME: is there useful debug info available here?
3455 TargetLowering::CallLoweringInfo CLI(DAG);
3456 CLI.setDebugLoc(dl).setChain(Chain).setLibCallee(
3458 DAG.getExternalSymbol("__tls_get_addr", PtrVT), std::move(Args));
3459
3460 std::pair<SDValue, SDValue> CallResult = LowerCallTo(CLI);
3461 return CallResult.first;
3462}
3463
3464// Lower ISD::GlobalTLSAddress using the "initial exec" or
3465// "local exec" model.
3466SDValue
3467ARMTargetLowering::LowerToTLSExecModels(GlobalAddressSDNode *GA,
3468 SelectionDAG &DAG,
3469 TLSModel::Model model) const {
3470 const GlobalValue *GV = GA->getGlobal();
3471 SDLoc dl(GA);
3472 SDValue Offset;
3473 SDValue Chain = DAG.getEntryNode();
3474 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3475 // Get the Thread Pointer
3476 SDValue ThreadPointer = DAG.getNode(ARMISD::THREAD_POINTER, dl, PtrVT);
3477
3478 if (model == TLSModel::InitialExec) {
3480 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
3481 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
3482 // Initial exec model.
3483 unsigned char PCAdj = Subtarget->isThumb() ? 4 : 8;
3484 ARMConstantPoolValue *CPV =
3485 ARMConstantPoolConstant::Create(GA->getGlobal(), ARMPCLabelIndex,
3487 true);
3488 Offset = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3489 Offset = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Offset);
3490 Offset = DAG.getLoad(
3491 PtrVT, dl, Chain, Offset,
3493 Chain = Offset.getValue(1);
3494
3495 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32);
3496 Offset = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Offset, PICLabel);
3497
3498 Offset = DAG.getLoad(
3499 PtrVT, dl, Chain, Offset,
3501 } else {
3502 // local exec model
3503 assert(model == TLSModel::LocalExec);
3504 ARMConstantPoolValue *CPV =
3506 Offset = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3507 Offset = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, Offset);
3508 Offset = DAG.getLoad(
3509 PtrVT, dl, Chain, Offset,
3511 }
3512
3513 // The address of the thread local variable is the add of the thread
3514 // pointer with the offset of the variable.
3515 return DAG.getNode(ISD::ADD, dl, PtrVT, ThreadPointer, Offset);
3516}
3517
3518SDValue
3519ARMTargetLowering::LowerGlobalTLSAddress(SDValue Op, SelectionDAG &DAG) const {
3520 GlobalAddressSDNode *GA = cast<GlobalAddressSDNode>(Op);
3521 if (DAG.getTarget().useEmulatedTLS())
3522 return LowerToTLSEmulatedModel(GA, DAG);
3523
3524 const Triple &TT = getTargetMachine().getTargetTriple();
3525 if (TT.isOSDarwin())
3526 return LowerGlobalTLSAddressDarwin(Op, DAG);
3527
3528 if (TT.isOSWindows())
3529 return LowerGlobalTLSAddressWindows(Op, DAG);
3530
3531 // TODO: implement the "local dynamic" model
3532 assert(TT.isOSBinFormatELF() && "Only ELF implemented here");
3534
3535 switch (model) {
3538 return LowerToTLSGeneralDynamicModel(GA, DAG);
3541 return LowerToTLSExecModels(GA, DAG, model);
3542 }
3543 llvm_unreachable("bogus TLS model");
3544}
3545
3546/// Return true if all users of V are within function F, looking through
3547/// ConstantExprs.
3548static bool allUsersAreInFunction(const Value *V, const Function *F) {
3549 SmallVector<const User*,4> Worklist(V->users());
3550 while (!Worklist.empty()) {
3551 auto *U = Worklist.pop_back_val();
3552 if (isa<ConstantExpr>(U)) {
3553 append_range(Worklist, U->users());
3554 continue;
3555 }
3556
3557 auto *I = dyn_cast<Instruction>(U);
3558 if (!I || I->getParent()->getParent() != F)
3559 return false;
3560 }
3561 return true;
3562}
3563
3565 const GlobalValue *GV, SelectionDAG &DAG,
3566 EVT PtrVT, const SDLoc &dl) {
3567 // If we're creating a pool entry for a constant global with unnamed address,
3568 // and the global is small enough, we can emit it inline into the constant pool
3569 // to save ourselves an indirection.
3570 //
3571 // This is a win if the constant is only used in one function (so it doesn't
3572 // need to be duplicated) or duplicating the constant wouldn't increase code
3573 // size (implying the constant is no larger than 4 bytes).
3574 const Function &F = DAG.getMachineFunction().getFunction();
3575
3576 // We rely on this decision to inline being idempotent and unrelated to the
3577 // use-site. We know that if we inline a variable at one use site, we'll
3578 // inline it elsewhere too (and reuse the constant pool entry). Fast-isel
3579 // doesn't know about this optimization, so bail out if it's enabled else
3580 // we could decide to inline here (and thus never emit the GV) but require
3581 // the GV from fast-isel generated code.
3584 return SDValue();
3585
3586 auto *GVar = dyn_cast<GlobalVariable>(GV);
3587 if (!GVar || !GVar->hasInitializer() ||
3588 !GVar->isConstant() || !GVar->hasGlobalUnnamedAddr() ||
3589 !GVar->hasLocalLinkage())
3590 return SDValue();
3591
3592 // If we inline a value that contains relocations, we move the relocations
3593 // from .data to .text. This is not allowed in position-independent code.
3594 auto *Init = GVar->getInitializer();
3595 if ((TLI->isPositionIndependent() || TLI->getSubtarget()->isROPI()) &&
3596 Init->needsDynamicRelocation())
3597 return SDValue();
3598
3599 // The constant islands pass can only really deal with alignment requests
3600 // <= 4 bytes and cannot pad constants itself. Therefore we cannot promote
3601 // any type wanting greater alignment requirements than 4 bytes. We also
3602 // can only promote constants that are multiples of 4 bytes in size or
3603 // are paddable to a multiple of 4. Currently we only try and pad constants
3604 // that are strings for simplicity.
3605 auto *CDAInit = dyn_cast<ConstantDataArray>(Init);
3606 unsigned Size = DAG.getDataLayout().getTypeAllocSize(Init->getType());
3607 Align PrefAlign = DAG.getDataLayout().getPreferredAlign(GVar);
3608 unsigned RequiredPadding = 4 - (Size % 4);
3609 bool PaddingPossible =
3610 RequiredPadding == 4 || (CDAInit && CDAInit->isString());
3611 if (!PaddingPossible || PrefAlign > 4 || Size > ConstpoolPromotionMaxSize ||
3612 Size == 0)
3613 return SDValue();
3614
3615 unsigned PaddedSize = Size + ((RequiredPadding == 4) ? 0 : RequiredPadding);
3617 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
3618
3619 // We can't bloat the constant pool too much, else the ConstantIslands pass
3620 // may fail to converge. If we haven't promoted this global yet (it may have
3621 // multiple uses), and promoting it would increase the constant pool size (Sz
3622 // > 4), ensure we have space to do so up to MaxTotal.
3623 if (!AFI->getGlobalsPromotedToConstantPool().count(GVar) && Size > 4)
3624 if (AFI->getPromotedConstpoolIncrease() + PaddedSize - 4 >=
3626 return SDValue();
3627
3628 // This is only valid if all users are in a single function; we can't clone
3629 // the constant in general. The LLVM IR unnamed_addr allows merging
3630 // constants, but not cloning them.
3631 //
3632 // We could potentially allow cloning if we could prove all uses of the
3633 // constant in the current function don't care about the address, like
3634 // printf format strings. But that isn't implemented for now.
3635 if (!allUsersAreInFunction(GVar, &F))
3636 return SDValue();
3637
3638 // We're going to inline this global. Pad it out if needed.
3639 if (RequiredPadding != 4) {
3640 StringRef S = CDAInit->getAsString();
3641
3643 std::copy(S.bytes_begin(), S.bytes_end(), V.begin());
3644 while (RequiredPadding--)
3645 V.push_back(0);
3647 }
3648
3649 auto CPVal = ARMConstantPoolConstant::Create(GVar, Init);
3650 SDValue CPAddr = DAG.getTargetConstantPool(CPVal, PtrVT, Align(4));
3651 if (!AFI->getGlobalsPromotedToConstantPool().count(GVar)) {
3654 PaddedSize - 4);
3655 }
3656 ++NumConstpoolPromoted;
3657 return DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
3658}
3659
3661 if (const GlobalAlias *GA = dyn_cast<GlobalAlias>(GV))
3662 if (!(GV = GA->getAliaseeObject()))
3663 return false;
3664 if (const auto *V = dyn_cast<GlobalVariable>(GV))
3665 return V->isConstant();
3666 return isa<Function>(GV);
3667}
3668
3669SDValue ARMTargetLowering::LowerGlobalAddress(SDValue Op,
3670 SelectionDAG &DAG) const {
3671 switch (Subtarget->getTargetTriple().getObjectFormat()) {
3672 default: llvm_unreachable("unknown object format");
3673 case Triple::COFF:
3674 return LowerGlobalAddressWindows(Op, DAG);
3675 case Triple::ELF:
3676 return LowerGlobalAddressELF(Op, DAG);
3677 case Triple::MachO:
3678 return LowerGlobalAddressDarwin(Op, DAG);
3679 }
3680}
3681
3682SDValue ARMTargetLowering::LowerGlobalAddressELF(SDValue Op,
3683 SelectionDAG &DAG) const {
3684 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3685 SDLoc dl(Op);
3686 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal();
3687 bool IsRO = isReadOnly(GV);
3688
3689 // promoteToConstantPool only if not generating XO text section
3690 if (GV->isDSOLocal() && !Subtarget->genExecuteOnly())
3691 if (SDValue V = promoteToConstantPool(this, GV, DAG, PtrVT, dl))
3692 return V;
3693
3694 if (isPositionIndependent()) {
3695 SDValue G = DAG.getTargetGlobalAddress(
3696 GV, dl, PtrVT, 0, GV->isDSOLocal() ? 0 : ARMII::MO_GOT);
3697 SDValue Result = DAG.getNode(ARMISD::WrapperPIC, dl, PtrVT, G);
3698 if (!GV->isDSOLocal())
3699 Result =
3700 DAG.getLoad(PtrVT, dl, DAG.getEntryNode(), Result,
3702 return Result;
3703 } else if (Subtarget->isROPI() && IsRO) {
3704 // PC-relative.
3705 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT);
3706 SDValue Result = DAG.getNode(ARMISD::WrapperPIC, dl, PtrVT, G);
3707 return Result;
3708 } else if (Subtarget->isRWPI() && !IsRO) {
3709 // SB-relative.
3710 SDValue RelAddr;
3711 if (Subtarget->useMovt()) {
3712 ++NumMovwMovt;
3713 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT, 0, ARMII::MO_SBREL);
3714 RelAddr = DAG.getNode(ARMISD::Wrapper, dl, PtrVT, G);
3715 } else { // use literal pool for address constant
3716 ARMConstantPoolValue *CPV =
3718 SDValue CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3719 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
3720 RelAddr = DAG.getLoad(
3721 PtrVT, dl, DAG.getEntryNode(), CPAddr,
3723 }
3724 SDValue SB = DAG.getCopyFromReg(DAG.getEntryNode(), dl, ARM::R9, PtrVT);
3725 SDValue Result = DAG.getNode(ISD::ADD, dl, PtrVT, SB, RelAddr);
3726 return Result;
3727 }
3728
3729 // If we have T2 ops, we can materialize the address directly via movt/movw
3730 // pair. This is always cheaper. If need to generate Execute Only code, and we
3731 // only have Thumb1 available, we can't use a constant pool and are forced to
3732 // use immediate relocations.
3733 if (Subtarget->useMovt() || Subtarget->genExecuteOnly()) {
3734 if (Subtarget->useMovt())
3735 ++NumMovwMovt;
3736 // FIXME: Once remat is capable of dealing with instructions with register
3737 // operands, expand this into two nodes.
3738 return DAG.getNode(ARMISD::Wrapper, dl, PtrVT,
3739 DAG.getTargetGlobalAddress(GV, dl, PtrVT));
3740 } else {
3741 SDValue CPAddr = DAG.getTargetConstantPool(GV, PtrVT, Align(4));
3742 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
3743 return DAG.getLoad(
3744 PtrVT, dl, DAG.getEntryNode(), CPAddr,
3746 }
3747}
3748
3749SDValue ARMTargetLowering::LowerGlobalAddressDarwin(SDValue Op,
3750 SelectionDAG &DAG) const {
3751 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() &&
3752 "ROPI/RWPI not currently supported for Darwin");
3753 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3754 SDLoc dl(Op);
3755 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal();
3756
3757 if (Subtarget->useMovt())
3758 ++NumMovwMovt;
3759
3760 // FIXME: Once remat is capable of dealing with instructions with register
3761 // operands, expand this into multiple nodes
3762 unsigned Wrapper =
3763 isPositionIndependent() ? ARMISD::WrapperPIC : ARMISD::Wrapper;
3764
3765 SDValue G = DAG.getTargetGlobalAddress(GV, dl, PtrVT, 0, ARMII::MO_NONLAZY);
3766 SDValue Result = DAG.getNode(Wrapper, dl, PtrVT, G);
3767
3768 if (Subtarget->isGVIndirectSymbol(GV))
3769 Result = DAG.getLoad(PtrVT, dl, DAG.getEntryNode(), Result,
3771 return Result;
3772}
3773
3774SDValue ARMTargetLowering::LowerGlobalAddressWindows(SDValue Op,
3775 SelectionDAG &DAG) const {
3776 assert(getTargetMachine().getTargetTriple().isOSWindows() &&
3777 "non-Windows COFF is not supported");
3778 assert(Subtarget->useMovt() &&
3779 "Windows on ARM expects to use movw/movt");
3780 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() &&
3781 "ROPI/RWPI not currently supported for Windows");
3782
3783 const TargetMachine &TM = getTargetMachine();
3784 const GlobalValue *GV = cast<GlobalAddressSDNode>(Op)->getGlobal();
3785 ARMII::TOF TargetFlags = ARMII::MO_NO_FLAG;
3786 if (GV->hasDLLImportStorageClass())
3787 TargetFlags = ARMII::MO_DLLIMPORT;
3788 else if (!TM.shouldAssumeDSOLocal(GV))
3789 TargetFlags = ARMII::MO_COFFSTUB;
3790 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3791 SDValue Result;
3792 SDLoc DL(Op);
3793
3794 ++NumMovwMovt;
3795
3796 // FIXME: Once remat is capable of dealing with instructions with register
3797 // operands, expand this into two nodes.
3798 Result = DAG.getNode(ARMISD::Wrapper, DL, PtrVT,
3799 DAG.getTargetGlobalAddress(GV, DL, PtrVT, /*offset=*/0,
3800 TargetFlags));
3801 if (TargetFlags & (ARMII::MO_DLLIMPORT | ARMII::MO_COFFSTUB))
3802 Result = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Result,
3804 return Result;
3805}
3806
3807SDValue
3808ARMTargetLowering::LowerEH_SJLJ_SETJMP(SDValue Op, SelectionDAG &DAG) const {
3809 SDLoc dl(Op);
3810 SDValue Val = DAG.getConstant(0, dl, MVT::i32);
3811 return DAG.getNode(ARMISD::EH_SJLJ_SETJMP, dl,
3812 DAG.getVTList(MVT::i32, MVT::Other), Op.getOperand(0),
3813 Op.getOperand(1), Val);
3814}
3815
3816SDValue
3817ARMTargetLowering::LowerEH_SJLJ_LONGJMP(SDValue Op, SelectionDAG &DAG) const {
3818 SDLoc dl(Op);
3819 return DAG.getNode(ARMISD::EH_SJLJ_LONGJMP, dl, MVT::Other, Op.getOperand(0),
3820 Op.getOperand(1), DAG.getConstant(0, dl, MVT::i32));
3821}
3822
3823SDValue ARMTargetLowering::LowerEH_SJLJ_SETUP_DISPATCH(SDValue Op,
3824 SelectionDAG &DAG) const {
3825 SDLoc dl(Op);
3826 return DAG.getNode(ARMISD::EH_SJLJ_SETUP_DISPATCH, dl, MVT::Other,
3827 Op.getOperand(0));
3828}
3829
3830SDValue ARMTargetLowering::LowerINTRINSIC_VOID(
3831 SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget) const {
3832 unsigned IntNo =
3833 Op.getConstantOperandVal(Op.getOperand(0).getValueType() == MVT::Other);
3834 switch (IntNo) {
3835 default:
3836 return SDValue(); // Don't custom lower most intrinsics.
3837 case Intrinsic::arm_gnu_eabi_mcount: {
3839 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3840 SDLoc dl(Op);
3841 SDValue Chain = Op.getOperand(0);
3842 // call "\01__gnu_mcount_nc"
3843 const ARMBaseRegisterInfo *ARI = Subtarget->getRegisterInfo();
3844 const uint32_t *Mask =
3846 assert(Mask && "Missing call preserved mask for calling convention");
3847 // Mark LR an implicit live-in.
3848 Register Reg = MF.addLiveIn(ARM::LR, getRegClassFor(MVT::i32));
3849 SDValue ReturnAddress =
3850 DAG.getCopyFromReg(DAG.getEntryNode(), dl, Reg, PtrVT);
3851 constexpr EVT ResultTys[] = {MVT::Other, MVT::Glue};
3852 SDValue Callee =
3853 DAG.getTargetExternalSymbol("\01__gnu_mcount_nc", PtrVT, 0);
3854 SDValue RegisterMask = DAG.getRegisterMask(Mask);
3855 if (Subtarget->isThumb())
3856 return SDValue(
3857 DAG.getMachineNode(
3858 ARM::tBL_PUSHLR, dl, ResultTys,
3859 {ReturnAddress, DAG.getTargetConstant(ARMCC::AL, dl, PtrVT),
3860 DAG.getRegister(0, PtrVT), Callee, RegisterMask, Chain}),
3861 0);
3862 return SDValue(
3863 DAG.getMachineNode(ARM::BL_PUSHLR, dl, ResultTys,
3864 {ReturnAddress, Callee, RegisterMask, Chain}),
3865 0);
3866 }
3867 }
3868}
3869
3870SDValue
3871ARMTargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op, SelectionDAG &DAG,
3872 const ARMSubtarget *Subtarget) const {
3873 unsigned IntNo = Op.getConstantOperandVal(0);
3874 SDLoc dl(Op);
3875 switch (IntNo) {
3876 default: return SDValue(); // Don't custom lower most intrinsics.
3877 case Intrinsic::localaddress: {
3878 const MachineFunction &MF = DAG.getMachineFunction();
3879 const auto *RegInfo = Subtarget->getRegisterInfo();
3880 unsigned Reg = RegInfo->getLocalAddressRegister(MF);
3881 return DAG.getCopyFromReg(DAG.getEntryNode(), dl, Reg,
3882 Op.getSimpleValueType());
3883 }
3884 case Intrinsic::eh_recoverfp: {
3885 SDValue FnOp = Op.getOperand(1);
3886 GlobalAddressSDNode *GSD = dyn_cast<GlobalAddressSDNode>(FnOp);
3887 auto *Fn = dyn_cast_or_null<Function>(GSD ? GSD->getGlobal() : nullptr);
3888 if (!Fn)
3890 "llvm.eh.recoverfp must take a function as the first argument");
3891 const auto *RegInfo = Subtarget->getRegisterInfo();
3892 Register BaseReg = RegInfo->getBaseRegister();
3894 MachineBasicBlock &MBB = *MF.begin();
3895 if (!MBB.isLiveIn(BaseReg))
3896 MBB.addLiveIn(BaseReg);
3897 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3898 return DAG.getCopyFromReg(DAG.getEntryNode(), dl, BaseReg, PtrVT);
3899 }
3900 case Intrinsic::thread_pointer: {
3901 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3902 return DAG.getNode(ARMISD::THREAD_POINTER, dl, PtrVT);
3903 }
3904 case Intrinsic::arm_cls: {
3905 // Note: arm_cls and arm_cls64 intrinsics are expanded directly here
3906 // in LowerINTRINSIC_WO_CHAIN since there's no native scalar CLS
3907 // instruction.
3908 const SDValue &Operand = Op.getOperand(1);
3909 const EVT VTy = Op.getValueType();
3910 return DAG.getNode(ISD::CTLS, dl, VTy, Operand);
3911 }
3912 case Intrinsic::arm_cls64: {
3913 // arm_cls64 returns i32 but takes i64 input.
3914 // Use ISD::CTLS for i64 and truncate the result.
3915 SDValue CTLS64 = DAG.getNode(ISD::CTLS, dl, MVT::i64, Op.getOperand(1));
3916 return DAG.getNode(ISD::TRUNCATE, dl, MVT::i32, CTLS64);
3917 }
3918 case Intrinsic::arm_neon_vcls:
3919 case Intrinsic::arm_mve_vcls: {
3920 // Lower vector CLS intrinsics to ISD::CTLS.
3921 // Vector CTLS is Legal when NEON/MVE is available (set elsewhere).
3922 const EVT VTy = Op.getValueType();
3923 return DAG.getNode(ISD::CTLS, dl, VTy, Op.getOperand(1));
3924 }
3925 case Intrinsic::eh_sjlj_lsda: {
3927 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
3928 unsigned ARMPCLabelIndex = AFI->createPICLabelUId();
3929 EVT PtrVT = getPointerTy(DAG.getDataLayout());
3930 SDValue CPAddr;
3931 bool IsPositionIndependent = isPositionIndependent();
3932 unsigned PCAdj = IsPositionIndependent ? (Subtarget->isThumb() ? 4 : 8) : 0;
3933 ARMConstantPoolValue *CPV =
3934 ARMConstantPoolConstant::Create(&MF.getFunction(), ARMPCLabelIndex,
3935 ARMCP::CPLSDA, PCAdj);
3936 CPAddr = DAG.getTargetConstantPool(CPV, PtrVT, Align(4));
3937 CPAddr = DAG.getNode(ARMISD::Wrapper, dl, MVT::i32, CPAddr);
3938 SDValue Result = DAG.getLoad(
3939 PtrVT, dl, DAG.getEntryNode(), CPAddr,
3941
3942 if (IsPositionIndependent) {
3943 SDValue PICLabel = DAG.getConstant(ARMPCLabelIndex, dl, MVT::i32);
3944 Result = DAG.getNode(ARMISD::PIC_ADD, dl, PtrVT, Result, PICLabel);
3945 }
3946 return Result;
3947 }
3948 case Intrinsic::arm_neon_vabs:
3949 return DAG.getNode(ISD::ABS, SDLoc(Op), Op.getValueType(),
3950 Op.getOperand(1));
3951 case Intrinsic::arm_neon_vabds:
3952 if (Op.getValueType().isInteger())
3953 return DAG.getNode(ISD::ABDS, SDLoc(Op), Op.getValueType(),
3954 Op.getOperand(1), Op.getOperand(2));
3955 return SDValue();
3956 case Intrinsic::arm_neon_vabdu:
3957 return DAG.getNode(ISD::ABDU, SDLoc(Op), Op.getValueType(),
3958 Op.getOperand(1), Op.getOperand(2));
3959 case Intrinsic::arm_neon_vmulls:
3960 case Intrinsic::arm_neon_vmullu: {
3961 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmulls)
3962 ? ARMISD::VMULLs : ARMISD::VMULLu;
3963 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(),
3964 Op.getOperand(1), Op.getOperand(2));
3965 }
3966 case Intrinsic::arm_neon_vminnm:
3967 case Intrinsic::arm_neon_vmaxnm: {
3968 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vminnm)
3969 ? ISD::FMINNUM : ISD::FMAXNUM;
3970 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(),
3971 Op.getOperand(1), Op.getOperand(2));
3972 }
3973 case Intrinsic::arm_neon_vminu:
3974 case Intrinsic::arm_neon_vmaxu: {
3975 if (Op.getValueType().isFloatingPoint())
3976 return SDValue();
3977 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vminu)
3978 ? ISD::UMIN : ISD::UMAX;
3979 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(),
3980 Op.getOperand(1), Op.getOperand(2));
3981 }
3982 case Intrinsic::arm_neon_vmins:
3983 case Intrinsic::arm_neon_vmaxs: {
3984 // v{min,max}s is overloaded between signed integers and floats.
3985 if (!Op.getValueType().isFloatingPoint()) {
3986 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmins)
3987 ? ISD::SMIN : ISD::SMAX;
3988 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(),
3989 Op.getOperand(1), Op.getOperand(2));
3990 }
3991 unsigned NewOpc = (IntNo == Intrinsic::arm_neon_vmins)
3992 ? ISD::FMINIMUM : ISD::FMAXIMUM;
3993 return DAG.getNode(NewOpc, SDLoc(Op), Op.getValueType(),
3994 Op.getOperand(1), Op.getOperand(2));
3995 }
3996 case Intrinsic::arm_neon_vtbl1:
3997 return DAG.getNode(ARMISD::VTBL1, SDLoc(Op), Op.getValueType(),
3998 Op.getOperand(1), Op.getOperand(2));
3999 case Intrinsic::arm_neon_vtbl2:
4000 return DAG.getNode(ARMISD::VTBL2, SDLoc(Op), Op.getValueType(),
4001 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
4002 case Intrinsic::arm_mve_pred_i2v:
4003 case Intrinsic::arm_mve_pred_v2i:
4004 return DAG.getNode(ARMISD::PREDICATE_CAST, SDLoc(Op), Op.getValueType(),
4005 Op.getOperand(1));
4006 case Intrinsic::arm_mve_vreinterpretq:
4007 return DAG.getNode(ARMISD::VECTOR_REG_CAST, SDLoc(Op), Op.getValueType(),
4008 Op.getOperand(1));
4009 case Intrinsic::arm_mve_lsll:
4010 return DAG.getNode(ARMISD::LSLL, SDLoc(Op), Op->getVTList(),
4011 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
4012 case Intrinsic::arm_mve_asrl:
4013 return DAG.getNode(ARMISD::ASRL, SDLoc(Op), Op->getVTList(),
4014 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
4015 case Intrinsic::arm_mve_vsli:
4016 return DAG.getNode(ARMISD::VSLIIMM, SDLoc(Op), Op->getVTList(),
4017 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
4018 case Intrinsic::arm_mve_vsri:
4019 return DAG.getNode(ARMISD::VSRIIMM, SDLoc(Op), Op->getVTList(),
4020 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
4021 }
4022}
4023
4025 const ARMSubtarget *Subtarget) {
4026 SDLoc dl(Op);
4027 auto SSID = static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
4028 if (SSID == SyncScope::SingleThread)
4029 return Op;
4030
4031 if (!Subtarget->hasDataBarrier()) {
4032 // Some ARMv6 cpus can support data barriers with an mcr instruction.
4033 // Thumb1 and pre-v6 ARM mode use a libcall instead and should never get
4034 // here.
4035 assert(Subtarget->hasV6Ops() && !Subtarget->isThumb() &&
4036 "Unexpected ISD::ATOMIC_FENCE encountered. Should be libcall!");
4037 return DAG.getNode(ARMISD::MEMBARRIER_MCR, dl, MVT::Other, Op.getOperand(0),
4038 DAG.getConstant(0, dl, MVT::i32));
4039 }
4040
4041 AtomicOrdering Ord =
4042 static_cast<AtomicOrdering>(Op.getConstantOperandVal(1));
4044 if (Subtarget->isMClass()) {
4045 // Only a full system barrier exists in the M-class architectures.
4047 } else if (Subtarget->preferISHSTBarriers() &&
4048 Ord == AtomicOrdering::Release) {
4049 // Swift happens to implement ISHST barriers in a way that's compatible with
4050 // Release semantics but weaker than ISH so we'd be fools not to use
4051 // it. Beware: other processors probably don't!
4053 }
4054
4055 return DAG.getNode(ISD::INTRINSIC_VOID, dl, MVT::Other, Op.getOperand(0),
4056 DAG.getConstant(Intrinsic::arm_dmb, dl, MVT::i32),
4057 DAG.getConstant(Domain, dl, MVT::i32));
4058}
4059
4061 const ARMSubtarget *Subtarget) {
4062 // ARM pre v5TE and Thumb1 does not have preload instructions.
4063 if (!(Subtarget->isThumb2() ||
4064 (!Subtarget->isThumb1Only() && Subtarget->hasV5TEOps())))
4065 // Just preserve the chain.
4066 return Op.getOperand(0);
4067
4068 SDLoc dl(Op);
4069 unsigned isRead = ~Op.getConstantOperandVal(2) & 1;
4070 if (!isRead &&
4071 (!Subtarget->hasV7Ops() || !Subtarget->hasMPExtension()))
4072 // ARMv7 with MP extension has PLDW.
4073 return Op.getOperand(0);
4074
4075 unsigned isData = Op.getConstantOperandVal(4);
4076 if (Subtarget->isThumb()) {
4077 // Invert the bits.
4078 isRead = ~isRead & 1;
4079 isData = ~isData & 1;
4080 }
4081
4082 return DAG.getNode(ARMISD::PRELOAD, dl, MVT::Other, Op.getOperand(0),
4083 Op.getOperand(1), DAG.getConstant(isRead, dl, MVT::i32),
4084 DAG.getConstant(isData, dl, MVT::i32));
4085}
4086
4089 ARMFunctionInfo *FuncInfo = MF.getInfo<ARMFunctionInfo>();
4090
4091 // vastart just stores the address of the VarArgsFrameIndex slot into the
4092 // memory location argument.
4093 SDLoc dl(Op);
4095 SDValue FR = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT);
4096 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4097 return DAG.getStore(Op.getOperand(0), dl, FR, Op.getOperand(1),
4098 MachinePointerInfo(SV));
4099}
4100
4101SDValue ARMTargetLowering::GetF64FormalArgument(CCValAssign &VA,
4102 CCValAssign &NextVA,
4103 SDValue &Root,
4104 SelectionDAG &DAG,
4105 const SDLoc &dl) const {
4107 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
4108
4109 const TargetRegisterClass *RC;
4110 if (AFI->isThumb1OnlyFunction())
4111 RC = &ARM::tGPRRegClass;
4112 else
4113 RC = &ARM::GPRRegClass;
4114
4115 // Transform the arguments stored in physical registers into virtual ones.
4116 Register Reg = MF.addLiveIn(VA.getLocReg(), RC);
4117 SDValue ArgValue = DAG.getCopyFromReg(Root, dl, Reg, MVT::i32);
4118
4119 SDValue ArgValue2;
4120 if (NextVA.isMemLoc()) {
4121 MachineFrameInfo &MFI = MF.getFrameInfo();
4122 int FI = MFI.CreateFixedObject(4, NextVA.getLocMemOffset(), true);
4123
4124 // Create load node to retrieve arguments from the stack.
4125 SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
4126 ArgValue2 = DAG.getLoad(
4127 MVT::i32, dl, Root, FIN,
4129 } else {
4130 Reg = MF.addLiveIn(NextVA.getLocReg(), RC);
4131 ArgValue2 = DAG.getCopyFromReg(Root, dl, Reg, MVT::i32);
4132 }
4133 if (!Subtarget->isLittle())
4134 std::swap (ArgValue, ArgValue2);
4135 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, ArgValue, ArgValue2);
4136}
4137
4138// The remaining GPRs hold either the beginning of variable-argument
4139// data, or the beginning of an aggregate passed by value (usually
4140// byval). Either way, we allocate stack slots adjacent to the data
4141// provided by our caller, and store the unallocated registers there.
4142// If this is a variadic function, the va_list pointer will begin with
4143// these values; otherwise, this reassembles a (byval) structure that
4144// was split between registers and memory.
4145// Return: The frame index registers were stored into.
4146int ARMTargetLowering::StoreByValRegs(CCState &CCInfo, SelectionDAG &DAG,
4147 const SDLoc &dl, SDValue &Chain,
4148 const Value *OrigArg,
4149 unsigned InRegsParamRecordIdx,
4150 int ArgOffset, unsigned ArgSize) const {
4151 // Currently, two use-cases possible:
4152 // Case #1. Non-var-args function, and we meet first byval parameter.
4153 // Setup first unallocated register as first byval register;
4154 // eat all remained registers
4155 // (these two actions are performed by HandleByVal method).
4156 // Then, here, we initialize stack frame with
4157 // "store-reg" instructions.
4158 // Case #2. Var-args function, that doesn't contain byval parameters.
4159 // The same: eat all remained unallocated registers,
4160 // initialize stack frame.
4161
4163 MachineFrameInfo &MFI = MF.getFrameInfo();
4164 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
4165 unsigned RBegin, REnd;
4166 if (InRegsParamRecordIdx < CCInfo.getInRegsParamsCount()) {
4167 CCInfo.getInRegsParamInfo(InRegsParamRecordIdx, RBegin, REnd);
4168 } else {
4169 unsigned RBeginIdx = CCInfo.getFirstUnallocated(GPRArgRegs);
4170 RBegin = RBeginIdx == 4 ? (unsigned)ARM::R4 : GPRArgRegs[RBeginIdx];
4171 REnd = ARM::R4;
4172 }
4173
4174 if (REnd != RBegin)
4175 ArgOffset = -4 * (ARM::R4 - RBegin);
4176
4177 auto PtrVT = getPointerTy(DAG.getDataLayout());
4178 int FrameIndex = MFI.CreateFixedObject(ArgSize, ArgOffset, false);
4179 SDValue FIN = DAG.getFrameIndex(FrameIndex, PtrVT);
4180
4182 const TargetRegisterClass *RC =
4183 AFI->isThumb1OnlyFunction() ? &ARM::tGPRRegClass : &ARM::GPRRegClass;
4184
4185 for (unsigned Reg = RBegin, i = 0; Reg < REnd; ++Reg, ++i) {
4186 Register VReg = MF.addLiveIn(Reg, RC);
4187 SDValue Val = DAG.getCopyFromReg(Chain, dl, VReg, MVT::i32);
4188 SDValue Store = DAG.getStore(Val.getValue(1), dl, Val, FIN,
4189 MachinePointerInfo(OrigArg, 4 * i));
4190 MemOps.push_back(Store);
4191 FIN = DAG.getNode(ISD::ADD, dl, PtrVT, FIN, DAG.getConstant(4, dl, PtrVT));
4192 }
4193
4194 if (!MemOps.empty())
4195 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOps);
4196 return FrameIndex;
4197}
4198
4199// Setup stack frame, the va_list pointer will start from.
4200void ARMTargetLowering::VarArgStyleRegisters(CCState &CCInfo, SelectionDAG &DAG,
4201 const SDLoc &dl, SDValue &Chain,
4202 unsigned ArgOffset,
4203 unsigned TotalArgRegsSaveSize,
4204 bool ForceMutable) const {
4206 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
4207
4208 // Try to store any remaining integer argument regs
4209 // to their spots on the stack so that they may be loaded by dereferencing
4210 // the result of va_next.
4211 // If there is no regs to be stored, just point address after last
4212 // argument passed via stack.
4213 int FrameIndex = StoreByValRegs(
4214 CCInfo, DAG, dl, Chain, nullptr, CCInfo.getInRegsParamsCount(),
4215 CCInfo.getStackSize(), std::max(4U, TotalArgRegsSaveSize));
4216 AFI->setVarArgsFrameIndex(FrameIndex);
4217}
4218
4219bool ARMTargetLowering::splitValueIntoRegisterParts(
4220 SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts,
4221 unsigned NumParts, MVT PartVT, std::optional<CallingConv::ID> CC) const {
4222 EVT ValueVT = Val.getValueType();
4223 if ((ValueVT == MVT::f16 || ValueVT == MVT::bf16) && PartVT == MVT::f32) {
4224 unsigned ValueBits = ValueVT.getSizeInBits();
4225 unsigned PartBits = PartVT.getSizeInBits();
4226 Val = DAG.getNode(ISD::BITCAST, DL, MVT::getIntegerVT(ValueBits), Val);
4227 Val = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::getIntegerVT(PartBits), Val);
4228 Val = DAG.getNode(ISD::BITCAST, DL, PartVT, Val);
4229 Parts[0] = Val;
4230 return true;
4231 }
4232 return false;
4233}
4234
4235SDValue ARMTargetLowering::joinRegisterPartsIntoValue(
4236 SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts,
4237 MVT PartVT, EVT ValueVT, std::optional<CallingConv::ID> CC) const {
4238 if ((ValueVT == MVT::f16 || ValueVT == MVT::bf16) && PartVT == MVT::f32) {
4239 unsigned ValueBits = ValueVT.getSizeInBits();
4240 unsigned PartBits = PartVT.getSizeInBits();
4241 SDValue Val = Parts[0];
4242
4243 Val = DAG.getNode(ISD::BITCAST, DL, MVT::getIntegerVT(PartBits), Val);
4244 Val = DAG.getNode(ISD::TRUNCATE, DL, MVT::getIntegerVT(ValueBits), Val);
4245 Val = DAG.getNode(ISD::BITCAST, DL, ValueVT, Val);
4246 return Val;
4247 }
4248 return SDValue();
4249}
4250
4251SDValue ARMTargetLowering::LowerFormalArguments(
4252 SDValue Chain, CallingConv::ID CallConv, bool isVarArg,
4253 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl,
4254 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
4256 MachineFrameInfo &MFI = MF.getFrameInfo();
4257
4258 ARMFunctionInfo *AFI = MF.getInfo<ARMFunctionInfo>();
4259
4260 // Assign locations to all of the incoming arguments.
4262 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs,
4263 *DAG.getContext());
4264 CCInfo.AnalyzeFormalArguments(Ins, CCAssignFnForCall(CallConv, isVarArg));
4265
4267 unsigned CurArgIdx = 0;
4268
4269 // Initially ArgRegsSaveSize is zero.
4270 // Then we increase this value each time we meet byval parameter.
4271 // We also increase this value in case of varargs function.
4272 AFI->setArgRegsSaveSize(0);
4273
4274 // Calculate the amount of stack space that we need to allocate to store
4275 // byval and variadic arguments that are passed in registers.
4276 // We need to know this before we allocate the first byval or variadic
4277 // argument, as they will be allocated a stack slot below the CFA (Canonical
4278 // Frame Address, the stack pointer at entry to the function).
4279 unsigned ArgRegBegin = ARM::R4;
4280 for (const CCValAssign &VA : ArgLocs) {
4281 if (CCInfo.getInRegsParamsProcessed() >= CCInfo.getInRegsParamsCount())
4282 break;
4283
4284 unsigned Index = VA.getValNo();
4285 ISD::ArgFlagsTy Flags = Ins[Index].Flags;
4286 if (!Flags.isByVal())
4287 continue;
4288
4289 assert(VA.isMemLoc() && "unexpected byval pointer in reg");
4290 unsigned RBegin, REnd;
4291 CCInfo.getInRegsParamInfo(CCInfo.getInRegsParamsProcessed(), RBegin, REnd);
4292 ArgRegBegin = std::min(ArgRegBegin, RBegin);
4293
4294 CCInfo.nextInRegsParam();
4295 }
4296 CCInfo.rewindByValRegsInfo();
4297
4298 int lastInsIndex = -1;
4299 if (isVarArg && MFI.hasVAStart()) {
4300 unsigned RegIdx = CCInfo.getFirstUnallocated(GPRArgRegs);
4301 if (RegIdx != std::size(GPRArgRegs))
4302 ArgRegBegin = std::min(ArgRegBegin, (unsigned)GPRArgRegs[RegIdx]);
4303 }
4304
4305 unsigned TotalArgRegsSaveSize = 4 * (ARM::R4 - ArgRegBegin);
4306 AFI->setArgRegsSaveSize(TotalArgRegsSaveSize);
4307 auto PtrVT = getPointerTy(DAG.getDataLayout());
4308
4309 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) {
4310 CCValAssign &VA = ArgLocs[i];
4311 if (Ins[VA.getValNo()].isOrigArg()) {
4312 std::advance(CurOrigArg,
4313 Ins[VA.getValNo()].getOrigArgIndex() - CurArgIdx);
4314 CurArgIdx = Ins[VA.getValNo()].getOrigArgIndex();
4315 }
4316 // Arguments stored in registers.
4317 if (VA.isRegLoc()) {
4318 EVT RegVT = VA.getLocVT();
4319 SDValue ArgValue;
4320
4321 if (VA.needsCustom() && VA.getLocVT() == MVT::v2f64) {
4322 // f64 and vector types are split up into multiple registers or
4323 // combinations of registers and stack slots.
4324 SDValue ArgValue1 =
4325 GetF64FormalArgument(VA, ArgLocs[++i], Chain, DAG, dl);
4326 VA = ArgLocs[++i]; // skip ahead to next loc
4327 SDValue ArgValue2;
4328 if (VA.isMemLoc()) {
4329 int FI = MFI.CreateFixedObject(8, VA.getLocMemOffset(), true);
4330 SDValue FIN = DAG.getFrameIndex(FI, PtrVT);
4331 ArgValue2 = DAG.getLoad(
4332 MVT::f64, dl, Chain, FIN,
4334 } else {
4335 ArgValue2 = GetF64FormalArgument(VA, ArgLocs[++i], Chain, DAG, dl);
4336 }
4337 ArgValue = DAG.getNode(ISD::UNDEF, dl, MVT::v2f64);
4338 ArgValue = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, ArgValue,
4339 ArgValue1, DAG.getIntPtrConstant(0, dl));
4340 ArgValue = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, ArgValue,
4341 ArgValue2, DAG.getIntPtrConstant(1, dl));
4342 } else if (VA.needsCustom() && VA.getLocVT() == MVT::f64) {
4343 ArgValue = GetF64FormalArgument(VA, ArgLocs[++i], Chain, DAG, dl);
4344 } else {
4345 const TargetRegisterClass *RC;
4346
4347 if (RegVT == MVT::f16 || RegVT == MVT::bf16)
4348 RC = &ARM::HPRRegClass;
4349 else if (RegVT == MVT::f32)
4350 RC = &ARM::SPRRegClass;
4351 else if (RegVT == MVT::f64 || RegVT == MVT::v4f16 ||
4352 RegVT == MVT::v4bf16)
4353 RC = &ARM::DPRRegClass;
4354 else if (RegVT == MVT::v2f64 || RegVT == MVT::v8f16 ||
4355 RegVT == MVT::v8bf16)
4356 RC = &ARM::QPRRegClass;
4357 else if (RegVT == MVT::i32)
4358 RC = AFI->isThumb1OnlyFunction() ? &ARM::tGPRRegClass
4359 : &ARM::GPRRegClass;
4360 else
4361 llvm_unreachable("RegVT not supported by FORMAL_ARGUMENTS Lowering");
4362
4363 // Transform the arguments in physical registers into virtual ones.
4364 Register Reg = MF.addLiveIn(VA.getLocReg(), RC);
4365 ArgValue = DAG.getCopyFromReg(Chain, dl, Reg, RegVT);
4366
4367 // If this value is passed in r0 and has the returned attribute (e.g.
4368 // C++ 'structors), record this fact for later use.
4369 if (VA.getLocReg() == ARM::R0 && Ins[VA.getValNo()].Flags.isReturned()) {
4370 AFI->setPreservesR0();
4371 }
4372 }
4373
4374 // If this is an 8 or 16-bit value, it is really passed promoted
4375 // to 32 bits. Insert an assert[sz]ext to capture this, then
4376 // truncate to the right size.
4377 switch (VA.getLocInfo()) {
4378 default: llvm_unreachable("Unknown loc info!");
4379 case CCValAssign::Full: break;
4380 case CCValAssign::BCvt:
4381 ArgValue = DAG.getNode(ISD::BITCAST, dl, VA.getValVT(), ArgValue);
4382 break;
4383 }
4384
4385 // f16 arguments have their size extended to 4 bytes and passed as if they
4386 // had been copied to the LSBs of a 32-bit register.
4387 // For that, it's passed extended to i32 (soft ABI) or to f32 (hard ABI)
4388 if (VA.needsCustom() &&
4389 (VA.getValVT() == MVT::f16 || VA.getValVT() == MVT::bf16))
4390 ArgValue = MoveToHPR(dl, DAG, VA.getLocVT(), VA.getValVT(), ArgValue);
4391
4392 // On CMSE Entry Functions, formal integer arguments whose bitwidth is
4393 // less than 32 bits must be sign- or zero-extended in the callee for
4394 // security reasons. Although the ABI mandates an extension done by the
4395 // caller, the latter cannot be trusted to follow the rules of the ABI.
4396 const ISD::InputArg &Arg = Ins[VA.getValNo()];
4397 if (AFI->isCmseNSEntryFunction() && Arg.ArgVT.isScalarInteger() &&
4398 RegVT.isScalarInteger() && Arg.ArgVT.bitsLT(MVT::i32))
4399 ArgValue = handleCMSEValue(ArgValue, Arg, DAG, dl);
4400
4401 InVals.push_back(ArgValue);
4402 } else { // VA.isRegLoc()
4403 // Only arguments passed on the stack should make it here.
4404 assert(VA.isMemLoc());
4405 assert(VA.getValVT() != MVT::i64 && "i64 should already be lowered");
4406
4407 int index = VA.getValNo();
4408
4409 // Some Ins[] entries become multiple ArgLoc[] entries.
4410 // Process them only once.
4411 if (index != lastInsIndex)
4412 {
4413 ISD::ArgFlagsTy Flags = Ins[index].Flags;
4414 // FIXME: For now, all byval parameter objects are marked mutable.
4415 // This can be changed with more analysis.
4416 // In case of tail call optimization mark all arguments mutable.
4417 // Since they could be overwritten by lowering of arguments in case of
4418 // a tail call.
4419 if (Flags.isByVal()) {
4420 assert(Ins[index].isOrigArg() &&
4421 "Byval arguments cannot be implicit");
4422 unsigned CurByValIndex = CCInfo.getInRegsParamsProcessed();
4423
4424 int FrameIndex = StoreByValRegs(
4425 CCInfo, DAG, dl, Chain, &*CurOrigArg, CurByValIndex,
4426 VA.getLocMemOffset(), Flags.getByValSize());
4427 InVals.push_back(DAG.getFrameIndex(FrameIndex, PtrVT));
4428 CCInfo.nextInRegsParam();
4429 } else if (VA.needsCustom() && (VA.getValVT() == MVT::f16 ||
4430 VA.getValVT() == MVT::bf16)) {
4431 // f16 and bf16 values are passed in the least-significant half of
4432 // a 4 byte stack slot. This is done as-if the extension was done
4433 // in a 32-bit register, so the actual bytes used for the value
4434 // differ between little and big endian.
4435 assert(VA.getLocVT().getSizeInBits() == 32);
4436 unsigned FIOffset = VA.getLocMemOffset();
4437 int FI = MFI.CreateFixedObject(VA.getLocVT().getSizeInBits() / 8,
4438 FIOffset, true);
4439
4440 SDValue Addr = DAG.getFrameIndex(FI, PtrVT);
4441 if (DAG.getDataLayout().isBigEndian())
4442 Addr = DAG.getObjectPtrOffset(dl, Addr, TypeSize::getFixed(2));
4443
4444 InVals.push_back(DAG.getLoad(VA.getValVT(), dl, Chain, Addr,
4446 DAG.getMachineFunction(), FI)));
4447
4448 } else {
4449 unsigned FIOffset = VA.getLocMemOffset();
4450 int FI = MFI.CreateFixedObject(VA.getLocVT().getSizeInBits()/8,
4451 FIOffset, true);
4452
4453 // Create load nodes to retrieve arguments from the stack.
4454 SDValue FIN = DAG.getFrameIndex(FI, PtrVT);
4455 InVals.push_back(DAG.getLoad(VA.getValVT(), dl, Chain, FIN,
4457 DAG.getMachineFunction(), FI)));
4458 }
4459 lastInsIndex = index;
4460 }
4461 }
4462 }
4463
4464 // varargs
4465 if (isVarArg && MFI.hasVAStart()) {
4466 VarArgStyleRegisters(CCInfo, DAG, dl, Chain, CCInfo.getStackSize(),
4467 TotalArgRegsSaveSize);
4468 if (AFI->isCmseNSEntryFunction()) {
4469 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
4471 "secure entry function must not be variadic", dl.getDebugLoc()));
4472 }
4473 }
4474
4475 unsigned StackArgSize = CCInfo.getStackSize();
4476 bool TailCallOpt = MF.getTarget().Options.GuaranteedTailCallOpt;
4477 if (canGuaranteeTCO(CallConv, TailCallOpt)) {
4478 // The only way to guarantee a tail call is if the callee restores its
4479 // argument area, but it must also keep the stack aligned when doing so.
4480 MaybeAlign StackAlign = DAG.getDataLayout().getStackAlignment();
4481 assert(StackAlign && "data layout string is missing stack alignment");
4482 StackArgSize = alignTo(StackArgSize, *StackAlign);
4483
4484 AFI->setArgumentStackToRestore(StackArgSize);
4485 }
4486 AFI->setArgumentStackSize(StackArgSize);
4487
4488 if (CCInfo.getStackSize() > 0 && AFI->isCmseNSEntryFunction()) {
4489 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
4491 "secure entry function requires arguments on stack", dl.getDebugLoc()));
4492 }
4493
4494 return Chain;
4495}
4496
4497/// isFloatingPointZero - Return true if this is +0.0.
4500 return CFP->getValueAPF().isPosZero();
4501 else if (ISD::isEXTLoad(Op.getNode()) || ISD::isNON_EXTLoad(Op.getNode())) {
4502 // Maybe this has already been legalized into the constant pool?
4503 if (Op.getOperand(1).getOpcode() == ARMISD::Wrapper) {
4504 SDValue WrapperOp = Op.getOperand(1).getOperand(0);
4506 if (const ConstantFP *CFP = dyn_cast<ConstantFP>(CP->getConstVal()))
4507 return CFP->getValueAPF().isPosZero();
4508 }
4509 } else if (Op->getOpcode() == ISD::BITCAST &&
4510 Op->getValueType(0) == MVT::f64) {
4511 // Handle (ISD::BITCAST (ARMISD::VMOVIMM (ISD::TargetConstant 0)) MVT::f64)
4512 // created by LowerConstantFP().
4513 SDValue BitcastOp = Op->getOperand(0);
4514 if (BitcastOp->getOpcode() == ARMISD::VMOVIMM &&
4515 isNullConstant(BitcastOp->getOperand(0)))
4516 return true;
4517 }
4518 return false;
4519}
4520
4522 // 0 - INT_MIN sign wraps, so no signed wrap means cmn is safe.
4523 if (Op->getFlags().hasNoSignedWrap())
4524 return true;
4525
4526 // We can still figure out if the second operand is safe to use
4527 // in a CMN instruction by checking if it is known to be not the minimum
4528 // signed value. If it is not, then we can safely use CMN.
4529 // Note: We can eventually remove this check and simply rely on
4530 // Op->getFlags().hasNoSignedWrap() once SelectionDAG/ISelLowering
4531 // consistently sets them appropriately when making said nodes.
4532
4533 KnownBits KnownSrc = DAG.computeKnownBits(Op.getOperand(1));
4534 return !KnownSrc.getSignedMinValue().isMinSignedValue();
4535}
4536
4538 return Op.getOpcode() == ISD::SUB && isNullConstant(Op.getOperand(0)) &&
4539 (isIntEqualitySetCC(CC) ||
4540 (isUnsignedIntSetCC(CC) && DAG.isKnownNeverZero(Op.getOperand(1))) ||
4541 (isSignedIntSetCC(CC) && isSafeSignedCMN(Op, DAG)));
4542}
4543
4544/// Returns how profitable it is to fold a comparison's operand's shift and/or
4545/// extension operations into the comparison instruction's second operand
4546/// (so_reg_imm / so_reg_reg for ARM, t2_so_reg for Thumb-2).
4548 // Thumb-1 CMP does not support shifted second operands.
4549 if (ST.isThumb1Only() || !Op.hasOneUse())
4550 return 0;
4551
4552 unsigned Opc = Op.getOpcode();
4553 if (Opc == ISD::SHL || Opc == ISD::SRL || Opc == ISD::SRA) {
4554 if (auto *ShiftAmt = dyn_cast<ConstantSDNode>(Op.getOperand(1)))
4555 return ShiftAmt->getZExtValue() <= 31 ? 1 : 0;
4556 // Register-controlled shift: only ARM-mode CMP/CMN (so_reg_reg) supports
4557 // this; Thumb-2 t2_so_reg requires an immediate shift amount.
4558 return ST.isThumb() ? 0 : 1;
4559 }
4560
4561 if (Opc == ISD::ROTR) {
4562 // Rotr constants will be normalized via mod 32, or & 31,
4563 // so we do not have to bounds check.
4564 if (isa<ConstantSDNode>(Op.getOperand(1)))
4565 return 1;
4566 return ST.isThumb() ? 0 : 1;
4567 }
4568
4569 return 0;
4570}
4571
4572/// Returns appropriate ARM CMP (cmp) and corresponding condition code for
4573/// the given operands.
4574SDValue ARMTargetLowering::getARMCmp(SDValue LHS, SDValue RHS, ISD::CondCode CC,
4575 SDValue &ARMcc, SelectionDAG &DAG,
4576 const SDLoc &dl) const {
4577 if (ConstantSDNode *RHSC = dyn_cast<ConstantSDNode>(RHS.getNode())) {
4578 unsigned C = RHSC->getZExtValue();
4579 if (!isLegalICmpImmediate((int32_t)C)) {
4580 // Constant does not fit, try adjusting it by one.
4581 switch (CC) {
4582 default: break;
4583 case ISD::SETLT:
4584 case ISD::SETGE:
4585 if (C != 0x80000000 && isLegalICmpImmediate(C-1)) {
4586 CC = (CC == ISD::SETLT) ? ISD::SETLE : ISD::SETGT;
4587 RHS = DAG.getConstant(C - 1, dl, MVT::i32);
4588 }
4589 break;
4590 case ISD::SETULT:
4591 case ISD::SETUGE:
4592 if (C != 0 && isLegalICmpImmediate(C-1)) {
4593 CC = (CC == ISD::SETULT) ? ISD::SETULE : ISD::SETUGT;
4594 RHS = DAG.getConstant(C - 1, dl, MVT::i32);
4595 }
4596 break;
4597 case ISD::SETLE:
4598 case ISD::SETGT:
4599 if (C != 0x7fffffff && isLegalICmpImmediate(C+1)) {
4600 CC = (CC == ISD::SETLE) ? ISD::SETLT : ISD::SETGE;
4601 RHS = DAG.getConstant(C + 1, dl, MVT::i32);
4602 }
4603 break;
4604 case ISD::SETULE:
4605 case ISD::SETUGT:
4606 if (C != 0xffffffff && isLegalICmpImmediate(C+1)) {
4607 CC = (CC == ISD::SETULE) ? ISD::SETULT : ISD::SETUGE;
4608 RHS = DAG.getConstant(C + 1, dl, MVT::i32);
4609 }
4610 break;
4611 }
4612 }
4613 }
4614
4615 // Thumb1 has very limited immediate modes, so turning an "and" into a
4616 // shift can save multiple instructions.
4617 //
4618 // If we have (x & C1), and C1 is an appropriate mask, we can transform it
4619 // into "((x << n) >> n)". But that isn't necessarily profitable on its
4620 // own. If it's the operand to an unsigned comparison with an immediate,
4621 // we can eliminate one of the shifts: we transform
4622 // "((x << n) >> n) == C2" to "(x << n) == (C2 << n)".
4623 //
4624 // We avoid transforming cases which aren't profitable due to encoding
4625 // details:
4626 //
4627 // 1. C2 fits into the immediate field of a cmp, and the transformed version
4628 // would not; in that case, we're essentially trading one immediate load for
4629 // another.
4630 // 2. C1 is 255 or 65535, so we can use uxtb or uxth.
4631 // 3. C2 is zero; we have other code for this special case.
4632 //
4633 // FIXME: Figure out profitability for Thumb2; we usually can't save an
4634 // instruction, since the AND is always one instruction anyway, but we could
4635 // use narrow instructions in some cases.
4636 if (Subtarget->isThumb1Only() && LHS->getOpcode() == ISD::AND &&
4637 LHS->hasOneUse() && isa<ConstantSDNode>(LHS.getOperand(1)) &&
4638 LHS.getValueType() == MVT::i32 && isa<ConstantSDNode>(RHS) &&
4639 !isSignedIntSetCC(CC)) {
4640 unsigned Mask = LHS.getConstantOperandVal(1);
4641 auto *RHSC = cast<ConstantSDNode>(RHS.getNode());
4642 uint64_t RHSV = RHSC->getZExtValue();
4643 if (isMask_32(Mask) && (RHSV & ~Mask) == 0 && Mask != 255 && Mask != 65535) {
4644 unsigned ShiftBits = llvm::countl_zero(Mask);
4645 if (RHSV && (RHSV > 255 || (RHSV << ShiftBits) <= 255)) {
4646 SDValue ShiftAmt = DAG.getConstant(ShiftBits, dl, MVT::i32);
4647 LHS = DAG.getNode(ISD::SHL, dl, MVT::i32, LHS.getOperand(0), ShiftAmt);
4648 RHS = DAG.getConstant(RHSV << ShiftBits, dl, MVT::i32);
4649 }
4650 }
4651 }
4652
4653 // The specific comparison "(x<<c) > 0x80000000U" can be optimized to a
4654 // single "lsls x, c+1". The shift sets the "C" and "Z" flags the same
4655 // way a cmp would.
4656 // FIXME: Add support for ARM/Thumb2; this would need isel patterns, and
4657 // some tweaks to the heuristics for the previous and->shift transform.
4658 // FIXME: Optimize cases where the LHS isn't a shift.
4659 if (Subtarget->isThumb1Only() && LHS->getOpcode() == ISD::SHL &&
4660 isa<ConstantSDNode>(RHS) && RHS->getAsZExtVal() == 0x80000000U &&
4661 CC == ISD::SETUGT && isa<ConstantSDNode>(LHS.getOperand(1)) &&
4662 LHS.getConstantOperandVal(1) < 31) {
4663 unsigned ShiftAmt = LHS.getConstantOperandVal(1) + 1;
4664 SDValue Shift =
4665 DAG.getNode(ARMISD::LSLS, dl, DAG.getVTList(MVT::i32, FlagsVT),
4666 LHS.getOperand(0), DAG.getConstant(ShiftAmt, dl, MVT::i32));
4667 ARMcc = DAG.getConstant(ARMCC::HI, dl, MVT::i32);
4668 return Shift.getValue(1);
4669 }
4670
4672
4673 unsigned CompareType;
4674 switch (CondCode) {
4675 default:
4676 CompareType = ARMISD::CMP;
4677 break;
4678 case ARMCC::EQ:
4679 case ARMCC::NE:
4680 // Uses only Z Flag
4681 CompareType = ARMISD::CMPZ;
4682 break;
4683 }
4684
4685 // TODO: Remove CMPZ check once we generalize and remove the CMPZ enum from
4686 // the codebase.
4687
4688 // TODO: When we have a solution to the vselect predicate not allowing pl/mi
4689 // all the time, allow those cases to be cmn too no matter what.
4690 if (CompareType != ARMISD::CMPZ && isCMN(RHS, CC, DAG)) {
4691 CompareType = ARMISD::CMN;
4692 RHS = RHS.getOperand(1);
4693 } else if (CompareType != ARMISD::CMPZ && isCMN(LHS, CC, DAG)) {
4694 CompareType = ARMISD::CMN;
4695 LHS = LHS.getOperand(1);
4697 }
4698
4699 // Prefer folding shifts / CMN into the cmp/cmn second operand (so_reg /
4700 // t2_so_reg). When both sides compete, pick the higher
4701 // getCmpOperandFoldingProfit. Only when RHS is not a legal icmp
4702 // immediate: otherwise keep the canonical (reg, imm) form.
4703 ConstantSDNode *C = dyn_cast<ConstantSDNode>(RHS.getNode());
4704 if (!C || !isLegalICmpImmediate(C->getSExtValue())) {
4705 if (getCmpOperandFoldingProfit(LHS, *Subtarget) >
4706 getCmpOperandFoldingProfit(RHS, *Subtarget)) {
4707 std::swap(LHS, RHS);
4708 if (CompareType == ARMISD::CMP)
4710 }
4711 }
4712
4713 // If the RHS is a constant zero then the V (overflow) flag will never be
4714 // set. This can allow us to simplify GE to PL or LT to MI, which can be
4715 // simpler for other passes (like the peephole optimiser) to deal with.
4716 if (isNullConstant(RHS)) {
4717 switch (CondCode) {
4718 default:
4719 break;
4720 case ARMCC::GE:
4722 break;
4723 case ARMCC::LT:
4725 break;
4726 }
4727 }
4728
4729 ARMcc = DAG.getConstant(CondCode, dl, MVT::i32);
4730 return DAG.getNode(CompareType, dl, FlagsVT, LHS, RHS);
4731}
4732
4733/// Returns a appropriate VFP CMP (fcmp{s|d}+fmstat) for the given operands.
4734SDValue ARMTargetLowering::getVFPCmp(SDValue LHS, SDValue RHS,
4735 SelectionDAG &DAG, const SDLoc &dl,
4736 bool Signaling) const {
4737 assert(Subtarget->hasFP64() || RHS.getValueType() != MVT::f64);
4738 SDValue Flags;
4740 Flags = DAG.getNode(Signaling ? ARMISD::CMPFPE : ARMISD::CMPFP, dl, FlagsVT,
4741 LHS, RHS);
4742 else
4743 Flags = DAG.getNode(Signaling ? ARMISD::CMPFPEw0 : ARMISD::CMPFPw0, dl,
4744 FlagsVT, LHS);
4745 return DAG.getNode(ARMISD::FMSTAT, dl, FlagsVT, Flags);
4746}
4747
4748// This function returns three things: the arithmetic computation itself
4749// (Value), a comparison (OverflowCmp), and a condition code (ARMcc). The
4750// comparison and the condition code define the case in which the arithmetic
4751// computation *does not* overflow.
4752std::pair<SDValue, SDValue>
4753ARMTargetLowering::getARMXALUOOp(SDValue Op, SelectionDAG &DAG,
4754 SDValue &ARMcc) const {
4755 assert(Op.getValueType() == MVT::i32 && "Unsupported value type");
4756
4757 SDValue Value, OverflowCmp;
4758 SDValue LHS = Op.getOperand(0);
4759 SDValue RHS = Op.getOperand(1);
4760 SDLoc dl(Op);
4761
4762 // FIXME: We are currently always generating CMPs because we don't support
4763 // generating CMN through the backend. This is not as good as the natural
4764 // CMP case because it causes a register dependency and cannot be folded
4765 // later.
4766
4767 switch (Op.getOpcode()) {
4768 default:
4769 llvm_unreachable("Unknown overflow instruction!");
4770 case ISD::SADDO:
4771 ARMcc = DAG.getConstant(ARMCC::VC, dl, MVT::i32);
4772 Value = DAG.getNode(ISD::ADD, dl, Op.getValueType(), LHS, RHS);
4773 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, FlagsVT, Value, LHS);
4774 break;
4775 case ISD::UADDO:
4776 ARMcc = DAG.getConstant(ARMCC::HS, dl, MVT::i32);
4777 // We use ADDC here to correspond to its use in LowerALUO.
4778 // We do not use it in the USUBO case as Value may not be used.
4779 Value = DAG.getNode(ARMISD::ADDC, dl,
4780 DAG.getVTList(Op.getValueType(), MVT::i32), LHS, RHS)
4781 .getValue(0);
4782 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, FlagsVT, Value, LHS);
4783 break;
4784 case ISD::SSUBO:
4785 ARMcc = DAG.getConstant(ARMCC::VC, dl, MVT::i32);
4786 Value = DAG.getNode(ISD::SUB, dl, Op.getValueType(), LHS, RHS);
4787 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, FlagsVT, LHS, RHS);
4788 break;
4789 case ISD::USUBO:
4790 ARMcc = DAG.getConstant(ARMCC::HS, dl, MVT::i32);
4791 Value = DAG.getNode(ISD::SUB, dl, Op.getValueType(), LHS, RHS);
4792 OverflowCmp = DAG.getNode(ARMISD::CMP, dl, FlagsVT, LHS, RHS);
4793 break;
4794 case ISD::UMULO:
4795 // We generate a UMUL_LOHI and then check if the high word is 0.
4796 ARMcc = DAG.getConstant(ARMCC::EQ, dl, MVT::i32);
4797 Value = DAG.getNode(ISD::UMUL_LOHI, dl,
4798 DAG.getVTList(Op.getValueType(), Op.getValueType()),
4799 LHS, RHS);
4800 OverflowCmp = DAG.getNode(ARMISD::CMPZ, dl, FlagsVT, Value.getValue(1),
4801 DAG.getConstant(0, dl, MVT::i32));
4802 Value = Value.getValue(0); // We only want the low 32 bits for the result.
4803 break;
4804 case ISD::SMULO:
4805 // We generate a SMUL_LOHI and then check if all the bits of the high word
4806 // are the same as the sign bit of the low word.
4807 ARMcc = DAG.getConstant(ARMCC::EQ, dl, MVT::i32);
4808 Value = DAG.getNode(ISD::SMUL_LOHI, dl,
4809 DAG.getVTList(Op.getValueType(), Op.getValueType()),
4810 LHS, RHS);
4811 OverflowCmp = DAG.getNode(ARMISD::CMPZ, dl, FlagsVT, Value.getValue(1),
4812 DAG.getNode(ISD::SRA, dl, Op.getValueType(),
4813 Value.getValue(0),
4814 DAG.getConstant(31, dl, MVT::i32)));
4815 Value = Value.getValue(0); // We only want the low 32 bits for the result.
4816 break;
4817 } // switch (...)
4818
4819 return std::make_pair(Value, OverflowCmp);
4820}
4821
4823 SDLoc DL(Value);
4824 EVT VT = Value.getValueType();
4825
4826 if (Invert)
4827 Value = DAG.getNode(ISD::SUB, DL, MVT::i32,
4828 DAG.getConstant(1, DL, MVT::i32), Value);
4829
4830 SDValue Cmp = DAG.getNode(ARMISD::SUBC, DL, DAG.getVTList(VT, MVT::i32),
4831 Value, DAG.getConstant(1, DL, VT));
4832 return Cmp.getValue(1);
4833}
4834
4836 bool Invert) {
4837 SDLoc DL(Flags);
4838
4839 if (Invert) {
4840 // Convert flags to boolean with ADDE 0,0,Carry then compute 1 - bool.
4841 SDValue BoolCarry = DAG.getNode(
4842 ARMISD::ADDE, DL, DAG.getVTList(VT, MVT::i32),
4843 DAG.getConstant(0, DL, VT), DAG.getConstant(0, DL, VT), Flags);
4844 return DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(1, DL, VT), BoolCarry);
4845 }
4846
4847 // Now convert the carry flag into a boolean carry. We do this
4848 // using ARMISD::ADDE 0, 0, Carry
4849 return DAG.getNode(ARMISD::ADDE, DL, DAG.getVTList(VT, MVT::i32),
4850 DAG.getConstant(0, DL, VT), DAG.getConstant(0, DL, VT),
4851 Flags);
4852}
4853
4854// Value is 1 if 'V' bit is 1, else 0
4856 SDLoc DL(Flags);
4857 SDValue Zero = DAG.getConstant(0, DL, VT);
4858 SDValue One = DAG.getConstant(1, DL, VT);
4859 SDValue ARMcc = DAG.getConstant(ARMCC::VS, DL, MVT::i32);
4860 return DAG.getNode(ARMISD::CMOV, DL, VT, Zero, One, ARMcc, Flags);
4861}
4862
4863SDValue ARMTargetLowering::LowerALUO(SDValue Op, SelectionDAG &DAG) const {
4864 // Let legalize expand this if it isn't a legal type yet.
4865 if (!isTypeLegal(Op.getValueType()))
4866 return SDValue();
4867
4868 SDValue LHS = Op.getOperand(0);
4869 SDValue RHS = Op.getOperand(1);
4870 SDLoc dl(Op);
4871
4872 EVT VT = Op.getValueType();
4873 SDVTList VTs = DAG.getVTList(VT, MVT::i32);
4874 SDValue Value;
4875 SDValue Overflow;
4876 switch (Op.getOpcode()) {
4877 case ISD::UADDO:
4878 Value = DAG.getNode(ARMISD::ADDC, dl, VTs, LHS, RHS);
4879 // Convert the carry flag into a boolean value.
4880 Overflow = carryFlagToValue(Value.getValue(1), VT, DAG, false);
4881 break;
4882 case ISD::USUBO:
4883 Value = DAG.getNode(ARMISD::SUBC, dl, VTs, LHS, RHS);
4884 // Convert the carry flag into a boolean value.
4885 Overflow = carryFlagToValue(Value.getValue(1), VT, DAG, true);
4886 break;
4887 default: {
4888 // Handle other operations with getARMXALUOOp
4889 SDValue OverflowCmp, ARMcc;
4890 std::tie(Value, OverflowCmp) = getARMXALUOOp(Op, DAG, ARMcc);
4891 // We use 0 and 1 as false and true values.
4892 // ARMcc represents the "no overflow" condition (e.g., VC for signed ops).
4893 // CMOV operand order is (FalseVal, TrueVal), so we put 1 in FalseVal
4894 // position to get Overflow=1 when the "no overflow" condition is false.
4895 Overflow =
4896 DAG.getNode(ARMISD::CMOV, dl, MVT::i32,
4897 DAG.getConstant(1, dl, MVT::i32), // FalseVal: overflow
4898 DAG.getConstant(0, dl, MVT::i32), // TrueVal: no overflow
4899 ARMcc, OverflowCmp);
4900 break;
4901 }
4902 }
4903
4904 return DAG.getNode(ISD::MERGE_VALUES, dl, VTs, Value, Overflow);
4905}
4906
4908 const ARMSubtarget *Subtarget) {
4909 EVT VT = Op.getValueType();
4910 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP() || Subtarget->isThumb1Only())
4911 return SDValue();
4912 if (!VT.isSimple())
4913 return SDValue();
4914
4915 unsigned NewOpcode;
4916 switch (VT.getSimpleVT().SimpleTy) {
4917 default:
4918 return SDValue();
4919 case MVT::i8:
4920 switch (Op->getOpcode()) {
4921 case ISD::UADDSAT:
4922 NewOpcode = ARMISD::UQADD8b;
4923 break;
4924 case ISD::SADDSAT:
4925 NewOpcode = ARMISD::QADD8b;
4926 break;
4927 case ISD::USUBSAT:
4928 NewOpcode = ARMISD::UQSUB8b;
4929 break;
4930 case ISD::SSUBSAT:
4931 NewOpcode = ARMISD::QSUB8b;
4932 break;
4933 }
4934 break;
4935 case MVT::i16:
4936 switch (Op->getOpcode()) {
4937 case ISD::UADDSAT:
4938 NewOpcode = ARMISD::UQADD16b;
4939 break;
4940 case ISD::SADDSAT:
4941 NewOpcode = ARMISD::QADD16b;
4942 break;
4943 case ISD::USUBSAT:
4944 NewOpcode = ARMISD::UQSUB16b;
4945 break;
4946 case ISD::SSUBSAT:
4947 NewOpcode = ARMISD::QSUB16b;
4948 break;
4949 }
4950 break;
4951 }
4952
4953 SDLoc dl(Op);
4954 SDValue Add =
4955 DAG.getNode(NewOpcode, dl, MVT::i32,
4956 DAG.getSExtOrTrunc(Op->getOperand(0), dl, MVT::i32),
4957 DAG.getSExtOrTrunc(Op->getOperand(1), dl, MVT::i32));
4958 return DAG.getNode(ISD::TRUNCATE, dl, VT, Add);
4959}
4960
4961SDValue ARMTargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
4962 SDValue Cond = Op.getOperand(0);
4963 SDValue SelectTrue = Op.getOperand(1);
4964 SDValue SelectFalse = Op.getOperand(2);
4965 SDLoc dl(Op);
4966 unsigned Opc = Cond.getOpcode();
4967
4968 if (Cond.getResNo() == 1 &&
4969 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO ||
4970 Opc == ISD::USUBO)) {
4971 if (!isTypeLegal(Cond->getValueType(0)))
4972 return SDValue();
4973
4974 SDValue Value, OverflowCmp;
4975 SDValue ARMcc;
4976 std::tie(Value, OverflowCmp) = getARMXALUOOp(Cond, DAG, ARMcc);
4977 EVT VT = Op.getValueType();
4978
4979 return getCMOV(dl, VT, SelectTrue, SelectFalse, ARMcc, OverflowCmp, DAG);
4980 }
4981
4982 // Convert:
4983 //
4984 // (select (cmov 1, 0, cond), t, f) -> (cmov t, f, cond)
4985 // (select (cmov 0, 1, cond), t, f) -> (cmov f, t, cond)
4986 //
4987 if (Cond.getOpcode() == ARMISD::CMOV && Cond.hasOneUse()) {
4988 const ConstantSDNode *CMOVTrue =
4989 dyn_cast<ConstantSDNode>(Cond.getOperand(0));
4990 const ConstantSDNode *CMOVFalse =
4991 dyn_cast<ConstantSDNode>(Cond.getOperand(1));
4992
4993 if (CMOVTrue && CMOVFalse) {
4994 unsigned CMOVTrueVal = CMOVTrue->getZExtValue();
4995 unsigned CMOVFalseVal = CMOVFalse->getZExtValue();
4996
4997 SDValue True;
4998 SDValue False;
4999 if (CMOVTrueVal == 1 && CMOVFalseVal == 0) {
5000 True = SelectTrue;
5001 False = SelectFalse;
5002 } else if (CMOVTrueVal == 0 && CMOVFalseVal == 1) {
5003 True = SelectFalse;
5004 False = SelectTrue;
5005 }
5006
5007 if (True.getNode() && False.getNode())
5008 return getCMOV(dl, Op.getValueType(), True, False, Cond.getOperand(2),
5009 Cond.getOperand(3), DAG);
5010 }
5011 }
5012
5013 return DAG.getSelectCC(dl, Cond,
5014 DAG.getConstant(0, dl, Cond.getValueType()),
5015 SelectTrue, SelectFalse, ISD::SETNE);
5016}
5017
5019 bool &swpCmpOps, bool &swpVselOps) {
5020 // Start by selecting the GE condition code for opcodes that return true for
5021 // 'equality'
5022 if (CC == ISD::SETUGE || CC == ISD::SETOGE || CC == ISD::SETOLE ||
5023 CC == ISD::SETULE || CC == ISD::SETGE || CC == ISD::SETLE)
5024 CondCode = ARMCC::GE;
5025
5026 // and GT for opcodes that return false for 'equality'.
5027 else if (CC == ISD::SETUGT || CC == ISD::SETOGT || CC == ISD::SETOLT ||
5028 CC == ISD::SETULT || CC == ISD::SETGT || CC == ISD::SETLT)
5029 CondCode = ARMCC::GT;
5030
5031 // Since we are constrained to GE/GT, if the opcode contains 'less', we need
5032 // to swap the compare operands.
5033 if (CC == ISD::SETOLE || CC == ISD::SETULE || CC == ISD::SETOLT ||
5034 CC == ISD::SETULT || CC == ISD::SETLE || CC == ISD::SETLT)
5035 swpCmpOps = true;
5036
5037 // Both GT and GE are ordered comparisons, and return false for 'unordered'.
5038 // If we have an unordered opcode, we need to swap the operands to the VSEL
5039 // instruction (effectively negating the condition).
5040 //
5041 // This also has the effect of swapping which one of 'less' or 'greater'
5042 // returns true, so we also swap the compare operands. It also switches
5043 // whether we return true for 'equality', so we compensate by picking the
5044 // opposite condition code to our original choice.
5045 if (CC == ISD::SETULE || CC == ISD::SETULT || CC == ISD::SETUGE ||
5046 CC == ISD::SETUGT) {
5047 swpCmpOps = !swpCmpOps;
5048 swpVselOps = !swpVselOps;
5049 CondCode = CondCode == ARMCC::GT ? ARMCC::GE : ARMCC::GT;
5050 }
5051
5052 // 'ordered' is 'anything but unordered', so use the VS condition code and
5053 // swap the VSEL operands.
5054 if (CC == ISD::SETO) {
5055 CondCode = ARMCC::VS;
5056 swpVselOps = true;
5057 }
5058
5059 // 'unordered or not equal' is 'anything but equal', so use the EQ condition
5060 // code and swap the VSEL operands. Also do this if we don't care about the
5061 // unordered case.
5062 if (CC == ISD::SETUNE || CC == ISD::SETNE) {
5063 CondCode = ARMCC::EQ;
5064 swpVselOps = true;
5065 }
5066}
5067
5068SDValue ARMTargetLowering::getCMOV(const SDLoc &dl, EVT VT, SDValue FalseVal,
5069 SDValue TrueVal, SDValue ARMcc,
5070 SDValue Flags, SelectionDAG &DAG) const {
5071 if (!Subtarget->hasFP64() && VT == MVT::f64) {
5072 FalseVal = DAG.getNode(ARMISD::VMOVRRD, dl,
5073 DAG.getVTList(MVT::i32, MVT::i32), FalseVal);
5074 TrueVal = DAG.getNode(ARMISD::VMOVRRD, dl,
5075 DAG.getVTList(MVT::i32, MVT::i32), TrueVal);
5076
5077 SDValue TrueLow = TrueVal.getValue(0);
5078 SDValue TrueHigh = TrueVal.getValue(1);
5079 SDValue FalseLow = FalseVal.getValue(0);
5080 SDValue FalseHigh = FalseVal.getValue(1);
5081
5082 SDValue Low = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, FalseLow, TrueLow,
5083 ARMcc, Flags);
5084 SDValue High = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, FalseHigh, TrueHigh,
5085 ARMcc, Flags);
5086
5087 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Low, High);
5088 }
5089 return DAG.getNode(ARMISD::CMOV, dl, VT, FalseVal, TrueVal, ARMcc, Flags);
5090}
5091
5092static bool isGTorGE(ISD::CondCode CC) {
5093 return CC == ISD::SETGT || CC == ISD::SETGE;
5094}
5095
5096static bool isLTorLE(ISD::CondCode CC) {
5097 return CC == ISD::SETLT || CC == ISD::SETLE;
5098}
5099
5100// See if a conditional (LHS CC RHS ? TrueVal : FalseVal) is lower-saturating.
5101// All of these conditions (and their <= and >= counterparts) will do:
5102// x < k ? k : x
5103// x > k ? x : k
5104// k < x ? x : k
5105// k > x ? k : x
5106static bool isLowerSaturate(const SDValue LHS, const SDValue RHS,
5107 const SDValue TrueVal, const SDValue FalseVal,
5108 const ISD::CondCode CC, const SDValue K) {
5109 return (isGTorGE(CC) &&
5110 ((K == LHS && K == TrueVal) || (K == RHS && K == FalseVal))) ||
5111 (isLTorLE(CC) &&
5112 ((K == RHS && K == TrueVal) || (K == LHS && K == FalseVal)));
5113}
5114
5115// Check if two chained conditionals could be converted into SSAT or USAT.
5116//
5117// SSAT can replace a set of two conditional selectors that bound a number to an
5118// interval of type [k, ~k] when k + 1 is a power of 2. Here are some examples:
5119//
5120// x < -k ? -k : (x > k ? k : x)
5121// x < -k ? -k : (x < k ? x : k)
5122// x > -k ? (x > k ? k : x) : -k
5123// x < k ? (x < -k ? -k : x) : k
5124// etc.
5125//
5126// LLVM canonicalizes these to either a min(max()) or a max(min())
5127// pattern. This function tries to match one of these and will return a SSAT
5128// node if successful.
5129//
5130// USAT works similarly to SSAT but bounds on the interval [0, k] where k + 1
5131// is a power of 2.
5133 EVT VT = Op.getValueType();
5134 SDValue V1 = Op.getOperand(0);
5135 SDValue K1 = Op.getOperand(1);
5136 SDValue TrueVal1 = Op.getOperand(2);
5137 SDValue FalseVal1 = Op.getOperand(3);
5138 ISD::CondCode CC1 = cast<CondCodeSDNode>(Op.getOperand(4))->get();
5139
5140 const SDValue Op2 = isa<ConstantSDNode>(TrueVal1) ? FalseVal1 : TrueVal1;
5141 if (Op2.getOpcode() != ISD::SELECT_CC)
5142 return SDValue();
5143
5144 SDValue V2 = Op2.getOperand(0);
5145 SDValue K2 = Op2.getOperand(1);
5146 SDValue TrueVal2 = Op2.getOperand(2);
5147 SDValue FalseVal2 = Op2.getOperand(3);
5148 ISD::CondCode CC2 = cast<CondCodeSDNode>(Op2.getOperand(4))->get();
5149
5150 SDValue V1Tmp = V1;
5151 SDValue V2Tmp = V2;
5152
5153 // Check that the registers and the constants match a max(min()) or min(max())
5154 // pattern
5155 if (V1Tmp != TrueVal1 || V2Tmp != TrueVal2 || K1 != FalseVal1 ||
5156 K2 != FalseVal2 ||
5157 !((isGTorGE(CC1) && isLTorLE(CC2)) || (isLTorLE(CC1) && isGTorGE(CC2))))
5158 return SDValue();
5159
5160 // Check that the constant in the lower-bound check is
5161 // the opposite of the constant in the upper-bound check
5162 // in 1's complement.
5164 return SDValue();
5165
5166 int64_t Val1 = cast<ConstantSDNode>(K1)->getSExtValue();
5167 int64_t Val2 = cast<ConstantSDNode>(K2)->getSExtValue();
5168 int64_t PosVal = std::max(Val1, Val2);
5169 int64_t NegVal = std::min(Val1, Val2);
5170
5171 if (!((Val1 > Val2 && isLTorLE(CC1)) || (Val1 < Val2 && isLTorLE(CC2))) ||
5172 !isPowerOf2_64(PosVal + 1))
5173 return SDValue();
5174
5175 // Handle the difference between USAT (unsigned) and SSAT (signed)
5176 // saturation
5177 // At this point, PosVal is guaranteed to be positive
5178 uint64_t K = PosVal;
5179 SDLoc dl(Op);
5180 if (Val1 == ~Val2)
5181 return DAG.getNode(ARMISD::SSAT, dl, VT, V2Tmp,
5182 DAG.getConstant(llvm::countr_one(K), dl, VT));
5183 if (NegVal == 0)
5184 return DAG.getNode(ARMISD::USAT, dl, VT, V2Tmp,
5185 DAG.getConstant(llvm::countr_one(K), dl, VT));
5186
5187 return SDValue();
5188}
5189
5190// Check if a condition of the type x < k ? k : x can be converted into a
5191// bit operation instead of conditional moves.
5192// Currently this is allowed given:
5193// - The conditions and values match up
5194// - k is 0 or -1 (all ones)
5195// This function will not check the last condition, thats up to the caller
5196// It returns true if the transformation can be made, and in such case
5197// returns x in V, and k in SatK.
5199 SDValue &SatK)
5200{
5201 SDValue LHS = Op.getOperand(0);
5202 SDValue RHS = Op.getOperand(1);
5203 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get();
5204 SDValue TrueVal = Op.getOperand(2);
5205 SDValue FalseVal = Op.getOperand(3);
5206
5208 ? &RHS
5209 : nullptr;
5210
5211 // No constant operation in comparison, early out
5212 if (!K)
5213 return false;
5214
5215 SDValue KTmp = isa<ConstantSDNode>(TrueVal) ? TrueVal : FalseVal;
5216 V = (KTmp == TrueVal) ? FalseVal : TrueVal;
5217 SDValue VTmp = (K && *K == LHS) ? RHS : LHS;
5218
5219 // If the constant on left and right side, or variable on left and right,
5220 // does not match, early out
5221 if (*K != KTmp || V != VTmp)
5222 return false;
5223
5224 if (isLowerSaturate(LHS, RHS, TrueVal, FalseVal, CC, *K)) {
5225 SatK = *K;
5226 return true;
5227 }
5228
5229 return false;
5230}
5231
5232bool ARMTargetLowering::isUnsupportedFloatingType(EVT VT) const {
5233 if (VT == MVT::f32)
5234 return !Subtarget->hasVFP2Base();
5235 if (VT == MVT::f64)
5236 return !Subtarget->hasFP64();
5237 if (VT == MVT::f16)
5238 return !Subtarget->hasFullFP16();
5239 return false;
5240}
5241
5242static SDValue matchCSET(unsigned &Opcode, bool &InvertCond, SDValue TrueVal,
5243 SDValue FalseVal, const ARMSubtarget *Subtarget) {
5244 ConstantSDNode *CFVal = dyn_cast<ConstantSDNode>(FalseVal);
5245 ConstantSDNode *CTVal = dyn_cast<ConstantSDNode>(TrueVal);
5246 if (!CFVal || !CTVal || !Subtarget->hasV8_1MMainlineOps())
5247 return SDValue();
5248
5249 unsigned TVal = CTVal->getZExtValue();
5250 unsigned FVal = CFVal->getZExtValue();
5251
5252 Opcode = 0;
5253 InvertCond = false;
5254 if (TVal == ~FVal) {
5255 Opcode = ARMISD::CSINV;
5256 } else if (TVal == ~FVal + 1) {
5257 Opcode = ARMISD::CSNEG;
5258 } else if (TVal + 1 == FVal) {
5259 Opcode = ARMISD::CSINC;
5260 } else if (TVal == FVal + 1) {
5261 Opcode = ARMISD::CSINC;
5262 std::swap(TrueVal, FalseVal);
5263 std::swap(TVal, FVal);
5264 InvertCond = !InvertCond;
5265 } else {
5266 return SDValue();
5267 }
5268
5269 // If one of the constants is cheaper than another, materialise the
5270 // cheaper one and let the csel generate the other.
5271 if (Opcode != ARMISD::CSINC &&
5272 HasLowerConstantMaterializationCost(FVal, TVal, Subtarget)) {
5273 std::swap(TrueVal, FalseVal);
5274 std::swap(TVal, FVal);
5275 InvertCond = !InvertCond;
5276 }
5277
5278 // Attempt to use ZR checking TVal is 0, possibly inverting the condition
5279 // to get there. CSINC not is invertable like the other two (~(~a) == a,
5280 // -(-a) == a, but (a+1)+1 != a).
5281 if (FVal == 0 && Opcode != ARMISD::CSINC) {
5282 std::swap(TrueVal, FalseVal);
5283 std::swap(TVal, FVal);
5284 InvertCond = !InvertCond;
5285 }
5286
5287 return TrueVal;
5288}
5289
5290SDValue ARMTargetLowering::LowerSELECT_CC(SDValue Op, SelectionDAG &DAG) const {
5291 EVT VT = Op.getValueType();
5292 SDLoc dl(Op);
5293
5294 // Try to convert two saturating conditional selects into a single SSAT
5295 if ((!Subtarget->isThumb() && Subtarget->hasV6Ops()) || Subtarget->isThumb2())
5296 if (SDValue SatValue = LowerSaturatingConditional(Op, DAG))
5297 return SatValue;
5298
5299 // Try to convert expressions of the form x < k ? k : x (and similar forms)
5300 // into more efficient bit operations, which is possible when k is 0 or -1
5301 // On ARM and Thumb-2 which have flexible operand 2 this will result in
5302 // single instructions. On Thumb the shift and the bit operation will be two
5303 // instructions.
5304 // Only allow this transformation on full-width (32-bit) operations
5305 SDValue LowerSatConstant;
5306 SDValue SatValue;
5307 if (VT == MVT::i32 &&
5308 isLowerSaturatingConditional(Op, SatValue, LowerSatConstant)) {
5309 SDValue ShiftV = DAG.getNode(ISD::SRA, dl, VT, SatValue,
5310 DAG.getConstant(31, dl, VT));
5311 if (isNullConstant(LowerSatConstant)) {
5312 SDValue NotShiftV = DAG.getNode(ISD::XOR, dl, VT, ShiftV,
5313 DAG.getAllOnesConstant(dl, VT));
5314 return DAG.getNode(ISD::AND, dl, VT, SatValue, NotShiftV);
5315 } else if (isAllOnesConstant(LowerSatConstant))
5316 return DAG.getNode(ISD::OR, dl, VT, SatValue, ShiftV);
5317 }
5318
5319 SDValue LHS = Op.getOperand(0);
5320 SDValue RHS = Op.getOperand(1);
5321 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get();
5322 SDValue TrueVal = Op.getOperand(2);
5323 SDValue FalseVal = Op.getOperand(3);
5324 ConstantSDNode *CFVal = dyn_cast<ConstantSDNode>(FalseVal);
5325 ConstantSDNode *RHSC = dyn_cast<ConstantSDNode>(RHS);
5326 if (Op.getValueType().isInteger()) {
5327
5328 // Check for SMAX(lhs, 0) and SMIN(lhs, 0) patterns.
5329 // (SELECT_CC setgt, lhs, 0, lhs, 0) -> (BIC lhs, (SRA lhs, typesize-1))
5330 // (SELECT_CC setlt, lhs, 0, lhs, 0) -> (AND lhs, (SRA lhs, typesize-1))
5331 // Both require less instructions than compare and conditional select.
5332 if ((CC == ISD::SETGT || CC == ISD::SETLT) && LHS == TrueVal && RHSC &&
5333 RHSC->isZero() && CFVal && CFVal->isZero() &&
5334 LHS.getValueType() == RHS.getValueType()) {
5335 EVT VT = LHS.getValueType();
5336 SDValue Shift =
5337 DAG.getNode(ISD::SRA, dl, VT, LHS,
5338 DAG.getConstant(VT.getSizeInBits() - 1, dl, VT));
5339
5340 if (CC == ISD::SETGT)
5341 Shift = DAG.getNOT(dl, Shift, VT);
5342
5343 return DAG.getNode(ISD::AND, dl, VT, LHS, Shift);
5344 }
5345
5346 // (SELECT_CC setlt, x, 0, 1, 0) -> SRL(x, bw-1)
5347 if (CC == ISD::SETLT && isNullConstant(RHS) && isOneConstant(TrueVal) &&
5348 isNullConstant(FalseVal) && LHS.getValueType() == VT)
5349 return DAG.getNode(ISD::SRL, dl, VT, LHS,
5350 DAG.getConstant(VT.getSizeInBits() - 1, dl, VT));
5351 }
5352
5353 if (LHS.getValueType() == MVT::i32) {
5354 unsigned Opcode;
5355 bool InvertCond;
5356 if (SDValue Op =
5357 matchCSET(Opcode, InvertCond, TrueVal, FalseVal, Subtarget)) {
5358 if (InvertCond)
5359 CC = ISD::getSetCCInverse(CC, LHS.getValueType());
5360
5361 SDValue ARMcc;
5362 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl);
5363 EVT VT = Op.getValueType();
5364 return DAG.getNode(Opcode, dl, VT, Op, Op, ARMcc, Cmp);
5365 }
5366 }
5367
5368 if (isUnsupportedFloatingType(LHS.getValueType())) {
5369 softenSetCCOperands(DAG, LHS.getValueType(), LHS, RHS, CC, dl, LHS, RHS);
5370
5371 // If softenSetCCOperands only returned one value, we should compare it to
5372 // zero.
5373 if (!RHS.getNode()) {
5374 RHS = DAG.getConstant(0, dl, LHS.getValueType());
5375 CC = ISD::SETNE;
5376 }
5377 }
5378
5379 if (LHS.getValueType() == MVT::i32) {
5380 // Try to generate VSEL on ARMv8.
5381 // The VSEL instruction can't use all the usual ARM condition
5382 // codes: it only has two bits to select the condition code, so it's
5383 // constrained to use only GE, GT, VS and EQ.
5384 //
5385 // To implement all the various ISD::SETXXX opcodes, we sometimes need to
5386 // swap the operands of the previous compare instruction (effectively
5387 // inverting the compare condition, swapping 'less' and 'greater') and
5388 // sometimes need to swap the operands to the VSEL (which inverts the
5389 // condition in the sense of firing whenever the previous condition didn't)
5390 if (Subtarget->hasFPARMv8Base() && (TrueVal.getValueType() == MVT::f16 ||
5391 TrueVal.getValueType() == MVT::f32 ||
5392 TrueVal.getValueType() == MVT::f64)) {
5394 if (CondCode == ARMCC::LT || CondCode == ARMCC::LE ||
5395 CondCode == ARMCC::VC || CondCode == ARMCC::NE) {
5396 CC = ISD::getSetCCInverse(CC, LHS.getValueType());
5397 std::swap(TrueVal, FalseVal);
5398 }
5399 }
5400
5401 SDValue ARMcc;
5402 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl);
5403 // Choose GE over PL, which vsel does now support
5404 if (ARMcc->getAsZExtVal() == ARMCC::PL)
5405 ARMcc = DAG.getConstant(ARMCC::GE, dl, MVT::i32);
5406 return getCMOV(dl, VT, FalseVal, TrueVal, ARMcc, Cmp, DAG);
5407 }
5408
5409 ARMCC::CondCodes CondCode, CondCode2;
5410 FPCCToARMCC(CC, CondCode, CondCode2);
5411
5412 // Normalize the fp compare. If RHS is zero we prefer to keep it there so we
5413 // match CMPFPw0 instead of CMPFP, though we don't do this for f16 because we
5414 // must use VSEL (limited condition codes), due to not having conditional f16
5415 // moves.
5416 if (Subtarget->hasFPARMv8Base() &&
5417 !(isFloatingPointZero(RHS) && TrueVal.getValueType() != MVT::f16) &&
5418 (TrueVal.getValueType() == MVT::f16 ||
5419 TrueVal.getValueType() == MVT::f32 ||
5420 TrueVal.getValueType() == MVT::f64)) {
5421 bool swpCmpOps = false;
5422 bool swpVselOps = false;
5423 checkVSELConstraints(CC, CondCode, swpCmpOps, swpVselOps);
5424
5425 if (CondCode == ARMCC::GT || CondCode == ARMCC::GE ||
5426 CondCode == ARMCC::VS || CondCode == ARMCC::EQ) {
5427 if (swpCmpOps)
5428 std::swap(LHS, RHS);
5429 if (swpVselOps)
5430 std::swap(TrueVal, FalseVal);
5431 }
5432 }
5433
5434 SDValue ARMcc = DAG.getConstant(CondCode, dl, MVT::i32);
5435 SDValue Cmp = getVFPCmp(LHS, RHS, DAG, dl);
5436 SDValue Result = getCMOV(dl, VT, FalseVal, TrueVal, ARMcc, Cmp, DAG);
5437 if (CondCode2 != ARMCC::AL) {
5438 SDValue ARMcc2 = DAG.getConstant(CondCode2, dl, MVT::i32);
5439 Result = getCMOV(dl, VT, Result, TrueVal, ARMcc2, Cmp, DAG);
5440 }
5441 return Result;
5442}
5443
5444/// canChangeToInt - Given the fp compare operand, return true if it is suitable
5445/// to morph to an integer compare sequence.
5446static bool canChangeToInt(SDValue Op, bool &SeenZero,
5447 const ARMSubtarget *Subtarget) {
5448 SDNode *N = Op.getNode();
5449 if (!N->hasOneUse())
5450 // Otherwise it requires moving the value from fp to integer registers.
5451 return false;
5452 if (!N->getNumValues())
5453 return false;
5454 EVT VT = Op.getValueType();
5455 if (VT != MVT::f32 && !Subtarget->isFPBrccSlow())
5456 // f32 case is generally profitable. f64 case only makes sense when vcmpe +
5457 // vmrs are very slow, e.g. cortex-a8.
5458 return false;
5459
5460 if (isFloatingPointZero(Op)) {
5461 SeenZero = true;
5462 return true;
5463 }
5464 return ISD::isNormalLoad(N);
5465}
5466
5469 return DAG.getConstant(0, SDLoc(Op), MVT::i32);
5470
5472 return DAG.getLoad(MVT::i32, SDLoc(Op), Ld->getChain(), Ld->getBasePtr(),
5473 Ld->getPointerInfo(), Ld->getAlign(),
5474 Ld->getMemOperand()->getFlags());
5475
5476 llvm_unreachable("Unknown VFP cmp argument!");
5477}
5478
5480 SDValue &RetVal1, SDValue &RetVal2) {
5481 SDLoc dl(Op);
5482
5483 if (isFloatingPointZero(Op)) {
5484 RetVal1 = DAG.getConstant(0, dl, MVT::i32);
5485 RetVal2 = DAG.getConstant(0, dl, MVT::i32);
5486 return;
5487 }
5488
5489 if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Op)) {
5490 SDValue Ptr = Ld->getBasePtr();
5491 RetVal1 =
5492 DAG.getLoad(MVT::i32, dl, Ld->getChain(), Ptr, Ld->getPointerInfo(),
5493 Ld->getAlign(), Ld->getMemOperand()->getFlags());
5494
5495 EVT PtrType = Ptr.getValueType();
5496 SDValue NewPtr = DAG.getNode(ISD::ADD, dl,
5497 PtrType, Ptr, DAG.getConstant(4, dl, PtrType));
5498 RetVal2 = DAG.getLoad(MVT::i32, dl, Ld->getChain(), NewPtr,
5499 Ld->getPointerInfo().getWithOffset(4),
5500 commonAlignment(Ld->getAlign(), 4),
5501 Ld->getMemOperand()->getFlags());
5502 return;
5503 }
5504
5505 llvm_unreachable("Unknown VFP cmp argument!");
5506}
5507
5508/// OptimizeVFPBrcond - With nnan and without daz, it's legal to optimize some
5509/// f32 and even f64 comparisons to integer ones.
5510SDValue
5511ARMTargetLowering::OptimizeVFPBrcond(SDValue Op, SelectionDAG &DAG) const {
5512 SDValue Chain = Op.getOperand(0);
5513 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get();
5514 SDValue LHS = Op.getOperand(2);
5515 SDValue RHS = Op.getOperand(3);
5516 SDValue Dest = Op.getOperand(4);
5517 SDLoc dl(Op);
5518
5519 bool LHSSeenZero = false;
5520 bool LHSOk = canChangeToInt(LHS, LHSSeenZero, Subtarget);
5521 bool RHSSeenZero = false;
5522 bool RHSOk = canChangeToInt(RHS, RHSSeenZero, Subtarget);
5523 if (LHSOk && RHSOk && (LHSSeenZero || RHSSeenZero)) {
5524 // If unsafe fp math optimization is enabled and there are no other uses of
5525 // the CMP operands, and the condition code is EQ or NE, we can optimize it
5526 // to an integer comparison.
5527 if (CC == ISD::SETOEQ)
5528 CC = ISD::SETEQ;
5529 else if (CC == ISD::SETUNE)
5530 CC = ISD::SETNE;
5531
5532 SDValue Mask = DAG.getConstant(0x7fffffff, dl, MVT::i32);
5533 SDValue ARMcc;
5534 if (LHS.getValueType() == MVT::f32) {
5535 LHS = DAG.getNode(ISD::AND, dl, MVT::i32,
5536 bitcastf32Toi32(LHS, DAG), Mask);
5537 RHS = DAG.getNode(ISD::AND, dl, MVT::i32,
5538 bitcastf32Toi32(RHS, DAG), Mask);
5539 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl);
5540 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc,
5541 Cmp);
5542 }
5543
5544 SDValue LHS1, LHS2;
5545 SDValue RHS1, RHS2;
5546 expandf64Toi32(LHS, DAG, LHS1, LHS2);
5547 expandf64Toi32(RHS, DAG, RHS1, RHS2);
5548 LHS2 = DAG.getNode(ISD::AND, dl, MVT::i32, LHS2, Mask);
5549 RHS2 = DAG.getNode(ISD::AND, dl, MVT::i32, RHS2, Mask);
5551 ARMcc = DAG.getConstant(CondCode, dl, MVT::i32);
5552 SDValue Ops[] = { Chain, ARMcc, LHS1, LHS2, RHS1, RHS2, Dest };
5553 return DAG.getNode(ARMISD::BCC_i64, dl, MVT::Other, Ops);
5554 }
5555
5556 return SDValue();
5557}
5558
5559// Generate CMP + CMOV for integer abs.
5560SDValue ARMTargetLowering::LowerABS(SDValue Op, SelectionDAG &DAG) const {
5561 SDLoc DL(Op);
5562
5563 SDValue Neg = DAG.getNegative(Op.getOperand(0), DL, MVT::i32);
5564
5565 // Generate CMP & CMOV.
5566 SDValue Cmp = DAG.getNode(ARMISD::CMP, DL, FlagsVT, Op.getOperand(0),
5567 DAG.getConstant(0, DL, MVT::i32));
5568 return DAG.getNode(ARMISD::CMOV, DL, MVT::i32, Op.getOperand(0), Neg,
5569 DAG.getConstant(ARMCC::MI, DL, MVT::i32), Cmp);
5570}
5571
5573 ARMCC::CondCodes CondCode =
5574 (ARMCC::CondCodes)cast<ConstantSDNode>(ARMcc)->getZExtValue();
5575 CondCode = ARMCC::getOppositeCondition(CondCode);
5576 return DAG.getConstant(CondCode, SDLoc(ARMcc), MVT::i32);
5577}
5578
5579SDValue ARMTargetLowering::LowerBRCOND(SDValue Op, SelectionDAG &DAG) const {
5580 SDValue Chain = Op.getOperand(0);
5581 SDValue Cond = Op.getOperand(1);
5582 SDValue Dest = Op.getOperand(2);
5583 SDLoc dl(Op);
5584
5585 // Optimize {s|u}{add|sub|mul}.with.overflow feeding into a branch
5586 // instruction.
5587 unsigned Opc = Cond.getOpcode();
5588 bool OptimizeMul = (Opc == ISD::SMULO || Opc == ISD::UMULO) &&
5589 !Subtarget->isThumb1Only();
5590 if (Cond.getResNo() == 1 &&
5591 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO ||
5592 Opc == ISD::USUBO || OptimizeMul)) {
5593 // Only lower legal XALUO ops.
5594 if (!isTypeLegal(Cond->getValueType(0)))
5595 return SDValue();
5596
5597 // The actual operation with overflow check.
5598 SDValue Value, OverflowCmp;
5599 SDValue ARMcc;
5600 std::tie(Value, OverflowCmp) = getARMXALUOOp(Cond, DAG, ARMcc);
5601
5602 // Reverse the condition code.
5603 ARMcc = getInvertedARMCondCode(ARMcc, DAG);
5604
5605 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc,
5606 OverflowCmp);
5607 }
5608
5609 return SDValue();
5610}
5611
5612SDValue ARMTargetLowering::LowerBR_CC(SDValue Op, SelectionDAG &DAG) const {
5613 SDValue Chain = Op.getOperand(0);
5614 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get();
5615 SDValue LHS = Op.getOperand(2);
5616 SDValue RHS = Op.getOperand(3);
5617 SDValue Dest = Op.getOperand(4);
5618 SDLoc dl(Op);
5619
5620 if (isUnsupportedFloatingType(LHS.getValueType())) {
5621 softenSetCCOperands(DAG, LHS.getValueType(), LHS, RHS, CC, dl, LHS, RHS);
5622
5623 // If softenSetCCOperands only returned one value, we should compare it to
5624 // zero.
5625 if (!RHS.getNode()) {
5626 RHS = DAG.getConstant(0, dl, LHS.getValueType());
5627 CC = ISD::SETNE;
5628 }
5629 }
5630
5631 // Optimize {s|u}{add|sub|mul}.with.overflow feeding into a branch
5632 // instruction.
5633 unsigned Opc = LHS.getOpcode();
5634 bool OptimizeMul = (Opc == ISD::SMULO || Opc == ISD::UMULO) &&
5635 !Subtarget->isThumb1Only();
5636 if (LHS.getResNo() == 1 && (isOneConstant(RHS) || isNullConstant(RHS)) &&
5637 (Opc == ISD::SADDO || Opc == ISD::UADDO || Opc == ISD::SSUBO ||
5638 Opc == ISD::USUBO || OptimizeMul) &&
5639 (CC == ISD::SETEQ || CC == ISD::SETNE)) {
5640 // Only lower legal XALUO ops.
5641 if (!isTypeLegal(LHS->getValueType(0)))
5642 return SDValue();
5643
5644 // The actual operation with overflow check.
5645 SDValue Value, OverflowCmp;
5646 SDValue ARMcc;
5647 std::tie(Value, OverflowCmp) = getARMXALUOOp(LHS.getValue(0), DAG, ARMcc);
5648
5649 if ((CC == ISD::SETNE) != isOneConstant(RHS)) {
5650 // Reverse the condition code.
5651 ARMcc = getInvertedARMCondCode(ARMcc, DAG);
5652 }
5653
5654 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc,
5655 OverflowCmp);
5656 }
5657
5658 if (LHS.getValueType() == MVT::i32) {
5659 SDValue ARMcc;
5660 SDValue Cmp = getARMCmp(LHS, RHS, CC, ARMcc, DAG, dl);
5661 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, Dest, ARMcc, Cmp);
5662 }
5663
5664 SDNodeFlags Flags = Op->getFlags();
5665 if (Flags.hasNoNaNs() &&
5666 DAG.getDenormalMode(MVT::f32) == DenormalMode::getIEEE() &&
5667 DAG.getDenormalMode(MVT::f64) == DenormalMode::getIEEE() &&
5668 (CC == ISD::SETEQ || CC == ISD::SETOEQ || CC == ISD::SETNE ||
5669 CC == ISD::SETUNE)) {
5670 if (SDValue Result = OptimizeVFPBrcond(Op, DAG))
5671 return Result;
5672 }
5673
5674 ARMCC::CondCodes CondCode, CondCode2;
5675 FPCCToARMCC(CC, CondCode, CondCode2);
5676
5677 SDValue ARMcc = DAG.getConstant(CondCode, dl, MVT::i32);
5678 SDValue Cmp = getVFPCmp(LHS, RHS, DAG, dl);
5679 SDValue Ops[] = {Chain, Dest, ARMcc, Cmp};
5680 SDValue Res = DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Ops);
5681 if (CondCode2 != ARMCC::AL) {
5682 ARMcc = DAG.getConstant(CondCode2, dl, MVT::i32);
5683 SDValue Ops[] = {Res, Dest, ARMcc, Cmp};
5684 Res = DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Ops);
5685 }
5686 return Res;
5687}
5688
5689SDValue ARMTargetLowering::LowerBR_JT(SDValue Op, SelectionDAG &DAG) const {
5690 SDValue Chain = Op.getOperand(0);
5691 SDValue Table = Op.getOperand(1);
5692 SDValue Index = Op.getOperand(2);
5693 SDLoc dl(Op);
5694
5695 EVT PTy = getPointerTy(DAG.getDataLayout());
5696 JumpTableSDNode *JT = cast<JumpTableSDNode>(Table);
5697 SDValue JTI = DAG.getTargetJumpTable(JT->getIndex(), PTy);
5698 Table = DAG.getNode(ARMISD::WrapperJT, dl, MVT::i32, JTI);
5699 Index = DAG.getNode(ISD::MUL, dl, PTy, Index, DAG.getConstant(4, dl, PTy));
5700 SDValue Addr = DAG.getNode(ISD::ADD, dl, PTy, Table, Index);
5701 if (Subtarget->isThumb2() || (Subtarget->hasV8MBaselineOps() && Subtarget->isThumb())) {
5702 // Thumb2 and ARMv8-M use a two-level jump. That is, it jumps into the jump table
5703 // which does another jump to the destination. This also makes it easier
5704 // to translate it to TBB / TBH later (Thumb2 only).
5705 // FIXME: This might not work if the function is extremely large.
5706 return DAG.getNode(ARMISD::BR2_JT, dl, MVT::Other, Chain,
5707 Addr, Op.getOperand(2), JTI);
5708 }
5709 if (isPositionIndependent() || Subtarget->isROPI()) {
5710 Addr =
5711 DAG.getLoad((EVT)MVT::i32, dl, Chain, Addr,
5713 Chain = Addr.getValue(1);
5714 Addr = DAG.getNode(ISD::ADD, dl, PTy, Table, Addr);
5715 return DAG.getNode(ARMISD::BR_JT, dl, MVT::Other, Chain, Addr, JTI);
5716 } else {
5717 Addr =
5718 DAG.getLoad(PTy, dl, Chain, Addr,
5720 Chain = Addr.getValue(1);
5721 return DAG.getNode(ARMISD::BR_JT, dl, MVT::Other, Chain, Addr, JTI);
5722 }
5723}
5724
5726 EVT VT = Op.getValueType();
5727 SDLoc dl(Op);
5728
5729 if (Op.getValueType().getVectorElementType() == MVT::i32) {
5730 if (Op.getOperand(0).getValueType().getVectorElementType() == MVT::f32)
5731 return Op;
5732 return DAG.UnrollVectorOp(Op.getNode());
5733 }
5734
5735 const bool HasFullFP16 = DAG.getSubtarget<ARMSubtarget>().hasFullFP16();
5736
5737 EVT NewTy;
5738 const EVT OpTy = Op.getOperand(0).getValueType();
5739 if (OpTy == MVT::v4f32)
5740 NewTy = MVT::v4i32;
5741 else if (OpTy == MVT::v4f16 && HasFullFP16)
5742 NewTy = MVT::v4i16;
5743 else if (OpTy == MVT::v8f16 && HasFullFP16)
5744 NewTy = MVT::v8i16;
5745 else
5746 llvm_unreachable("Invalid type for custom lowering!");
5747
5748 if (VT != MVT::v4i16 && VT != MVT::v8i16)
5749 return DAG.UnrollVectorOp(Op.getNode());
5750
5751 Op = DAG.getNode(Op.getOpcode(), dl, NewTy, Op.getOperand(0));
5752 return DAG.getNode(ISD::TRUNCATE, dl, VT, Op);
5753}
5754
5755SDValue ARMTargetLowering::LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const {
5756 EVT VT = Op.getValueType();
5757 if (VT.isVector())
5758 return LowerVectorFP_TO_INT(Op, DAG);
5759
5760 bool IsStrict = Op->isStrictFPOpcode();
5761 SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
5762
5763 if (isUnsupportedFloatingType(SrcVal.getValueType())) {
5764 RTLIB::Libcall LC;
5765 if (Op.getOpcode() == ISD::FP_TO_SINT ||
5766 Op.getOpcode() == ISD::STRICT_FP_TO_SINT)
5767 LC = RTLIB::getFPTOSINT(SrcVal.getValueType(),
5768 Op.getValueType());
5769 else
5770 LC = RTLIB::getFPTOUINT(SrcVal.getValueType(),
5771 Op.getValueType());
5772 SDLoc Loc(Op);
5773 MakeLibCallOptions CallOptions;
5774 SDValue Chain = IsStrict ? Op.getOperand(0) : SDValue();
5775 SDValue Result;
5776 std::tie(Result, Chain) = makeLibCall(DAG, LC, Op.getValueType(), SrcVal,
5777 CallOptions, Loc, Chain);
5778 return IsStrict ? DAG.getMergeValues({Result, Chain}, Loc) : Result;
5779 }
5780
5781 return Op;
5782}
5783
5785 const ARMSubtarget *Subtarget) {
5786 EVT VT = Op.getValueType();
5787 EVT ToVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
5788 EVT FromVT = Op.getOperand(0).getValueType();
5789
5790 if (VT == MVT::i32 && ToVT == MVT::i32 && FromVT == MVT::f32)
5791 return Op;
5792 if (VT == MVT::i32 && ToVT == MVT::i32 && FromVT == MVT::f64 &&
5793 Subtarget->hasFP64())
5794 return Op;
5795 if (VT == MVT::i32 && ToVT == MVT::i32 && FromVT == MVT::f16 &&
5796 Subtarget->hasFullFP16())
5797 return Op;
5798 if (VT == MVT::v4i32 && ToVT == MVT::i32 && FromVT == MVT::v4f32 &&
5799 Subtarget->hasMVEFloatOps())
5800 return Op;
5801 if (VT == MVT::v8i16 && ToVT == MVT::i16 && FromVT == MVT::v8f16 &&
5802 Subtarget->hasMVEFloatOps())
5803 return Op;
5804
5805 if (FromVT != MVT::v4f32 && FromVT != MVT::v8f16)
5806 return SDValue();
5807
5808 SDLoc DL(Op);
5809 bool IsSigned = Op.getOpcode() == ISD::FP_TO_SINT_SAT;
5810 unsigned BW = ToVT.getScalarSizeInBits() - IsSigned;
5811 SDValue CVT = DAG.getNode(Op.getOpcode(), DL, VT, Op.getOperand(0),
5812 DAG.getValueType(VT.getScalarType()));
5813 SDValue Max = DAG.getNode(IsSigned ? ISD::SMIN : ISD::UMIN, DL, VT, CVT,
5814 DAG.getConstant((1 << BW) - 1, DL, VT));
5815 if (IsSigned)
5816 Max = DAG.getNode(ISD::SMAX, DL, VT, Max,
5817 DAG.getSignedConstant(-(1 << BW), DL, VT));
5818 return Max;
5819}
5820
5822 EVT VT = Op.getValueType();
5823 SDLoc dl(Op);
5824
5825 if (Op.getOperand(0).getValueType().getVectorElementType() == MVT::i32) {
5826 if (VT.getVectorElementType() == MVT::f32)
5827 return Op;
5828 return DAG.UnrollVectorOp(Op.getNode());
5829 }
5830
5831 assert((Op.getOperand(0).getValueType() == MVT::v4i16 ||
5832 Op.getOperand(0).getValueType() == MVT::v8i16) &&
5833 "Invalid type for custom lowering!");
5834
5835 const bool HasFullFP16 = DAG.getSubtarget<ARMSubtarget>().hasFullFP16();
5836
5837 EVT DestVecType;
5838 if (VT == MVT::v4f32)
5839 DestVecType = MVT::v4i32;
5840 else if (VT == MVT::v4f16 && HasFullFP16)
5841 DestVecType = MVT::v4i16;
5842 else if (VT == MVT::v8f16 && HasFullFP16)
5843 DestVecType = MVT::v8i16;
5844 else
5845 return DAG.UnrollVectorOp(Op.getNode());
5846
5847 unsigned CastOpc;
5848 unsigned Opc;
5849 switch (Op.getOpcode()) {
5850 default: llvm_unreachable("Invalid opcode!");
5851 case ISD::SINT_TO_FP:
5852 CastOpc = ISD::SIGN_EXTEND;
5854 break;
5855 case ISD::UINT_TO_FP:
5856 CastOpc = ISD::ZERO_EXTEND;
5858 break;
5859 }
5860
5861 Op = DAG.getNode(CastOpc, dl, DestVecType, Op.getOperand(0));
5862 return DAG.getNode(Opc, dl, VT, Op);
5863}
5864
5865SDValue ARMTargetLowering::LowerINT_TO_FP(SDValue Op, SelectionDAG &DAG) const {
5866 EVT VT = Op.getValueType();
5867 if (VT.isVector())
5868 return LowerVectorINT_TO_FP(Op, DAG);
5869
5870 bool IsStrict = Op->isStrictFPOpcode();
5871 SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
5872
5873 if (isUnsupportedFloatingType(VT)) {
5874 RTLIB::Libcall LC;
5875 if (Op.getOpcode() == ISD::SINT_TO_FP ||
5876 Op.getOpcode() == ISD::STRICT_SINT_TO_FP)
5877 LC = RTLIB::getSINTTOFP(SrcVal.getValueType(), Op.getValueType());
5878 else
5879 LC = RTLIB::getUINTTOFP(SrcVal.getValueType(), Op.getValueType());
5880 SDLoc Loc(Op);
5881 MakeLibCallOptions CallOptions;
5882 SDValue Chain = IsStrict ? Op.getOperand(0) : SDValue();
5883 SDValue Result;
5884 std::tie(Result, Chain) = makeLibCall(DAG, LC, Op.getValueType(), SrcVal,
5885 CallOptions, Loc, Chain);
5886 return IsStrict ? DAG.getMergeValues({Result, Chain}, Loc) : Result;
5887 }
5888
5889 return Op;
5890}
5891
5892SDValue ARMTargetLowering::LowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const {
5893 // Implement fcopysign with a fabs and a conditional fneg.
5894 SDValue Tmp0 = Op.getOperand(0);
5895 SDValue Tmp1 = Op.getOperand(1);
5896 SDLoc dl(Op);
5897 EVT VT = Op.getValueType();
5898 EVT SrcVT = Tmp1.getValueType();
5899 bool InGPR = Tmp0.getOpcode() == ISD::BITCAST ||
5900 Tmp0.getOpcode() == ARMISD::VMOVDRR;
5901 bool UseNEON = !InGPR && Subtarget->hasNEON();
5902
5903 if (UseNEON) {
5904 // Use VBSL to copy the sign bit.
5905 unsigned EncodedVal = ARM_AM::createVMOVModImm(0x6, 0x80);
5906 SDValue Mask = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v2i32,
5907 DAG.getTargetConstant(EncodedVal, dl, MVT::i32));
5908 EVT OpVT = (VT == MVT::f32) ? MVT::v2i32 : MVT::v1i64;
5909 if (VT == MVT::f64)
5910 Mask = DAG.getNode(ARMISD::VSHLIMM, dl, OpVT,
5911 DAG.getNode(ISD::BITCAST, dl, OpVT, Mask),
5912 DAG.getConstant(32, dl, MVT::i32));
5913 else /*if (VT == MVT::f32)*/
5914 Tmp0 = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2f32, Tmp0);
5915 if (SrcVT == MVT::f32) {
5916 Tmp1 = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2f32, Tmp1);
5917 if (VT == MVT::f64)
5918 Tmp1 = DAG.getNode(ARMISD::VSHLIMM, dl, OpVT,
5919 DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp1),
5920 DAG.getConstant(32, dl, MVT::i32));
5921 } else if (VT == MVT::f32)
5922 Tmp1 = DAG.getNode(ARMISD::VSHRuIMM, dl, MVT::v1i64,
5923 DAG.getNode(ISD::BITCAST, dl, MVT::v1i64, Tmp1),
5924 DAG.getConstant(32, dl, MVT::i32));
5925 Tmp0 = DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp0);
5926 Tmp1 = DAG.getNode(ISD::BITCAST, dl, OpVT, Tmp1);
5927
5928 SDValue AllOnes = DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0xff),
5929 dl, MVT::i32);
5930 AllOnes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v8i8, AllOnes);
5931 SDValue MaskNot = DAG.getNode(ISD::XOR, dl, OpVT, Mask,
5932 DAG.getNode(ISD::BITCAST, dl, OpVT, AllOnes));
5933
5934 SDValue Res = DAG.getNode(ISD::OR, dl, OpVT,
5935 DAG.getNode(ISD::AND, dl, OpVT, Tmp1, Mask),
5936 DAG.getNode(ISD::AND, dl, OpVT, Tmp0, MaskNot));
5937 if (VT == MVT::f32) {
5938 Res = DAG.getNode(ISD::BITCAST, dl, MVT::v2f32, Res);
5939 Res = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f32, Res,
5940 DAG.getConstant(0, dl, MVT::i32));
5941 } else {
5942 Res = DAG.getNode(ISD::BITCAST, dl, MVT::f64, Res);
5943 }
5944
5945 return Res;
5946 }
5947
5948 // Bitcast operand 1 to i32.
5949 if (SrcVT == MVT::f64)
5950 Tmp1 = DAG.getNode(ARMISD::VMOVRRD, dl, DAG.getVTList(MVT::i32, MVT::i32),
5951 Tmp1).getValue(1);
5952 Tmp1 = DAG.getNode(ISD::BITCAST, dl, MVT::i32, Tmp1);
5953
5954 // Or in the signbit with integer operations.
5955 SDValue Mask1 = DAG.getConstant(0x80000000, dl, MVT::i32);
5956 SDValue Mask2 = DAG.getConstant(0x7fffffff, dl, MVT::i32);
5957 Tmp1 = DAG.getNode(ISD::AND, dl, MVT::i32, Tmp1, Mask1);
5958 if (VT == MVT::f32) {
5959 Tmp0 = DAG.getNode(ISD::AND, dl, MVT::i32,
5960 DAG.getNode(ISD::BITCAST, dl, MVT::i32, Tmp0), Mask2);
5961 return DAG.getNode(ISD::BITCAST, dl, MVT::f32,
5962 DAG.getNode(ISD::OR, dl, MVT::i32, Tmp0, Tmp1));
5963 }
5964
5965 // f64: Or the high part with signbit and then combine two parts.
5966 Tmp0 = DAG.getNode(ARMISD::VMOVRRD, dl, DAG.getVTList(MVT::i32, MVT::i32),
5967 Tmp0);
5968 SDValue Lo = Tmp0.getValue(0);
5969 SDValue Hi = DAG.getNode(ISD::AND, dl, MVT::i32, Tmp0.getValue(1), Mask2);
5970 Hi = DAG.getNode(ISD::OR, dl, MVT::i32, Hi, Tmp1);
5971 return DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi);
5972}
5973
5974SDValue ARMTargetLowering::LowerRETURNADDR(SDValue Op, SelectionDAG &DAG) const{
5976 MachineFrameInfo &MFI = MF.getFrameInfo();
5977 MFI.setReturnAddressIsTaken(true);
5978
5979 EVT VT = Op.getValueType();
5980 SDLoc dl(Op);
5981 unsigned Depth = Op.getConstantOperandVal(0);
5982 if (Depth) {
5983 SDValue FrameAddr = LowerFRAMEADDR(Op, DAG);
5984 SDValue Offset = DAG.getConstant(4, dl, MVT::i32);
5985 return DAG.getLoad(VT, dl, DAG.getEntryNode(),
5986 DAG.getNode(ISD::ADD, dl, VT, FrameAddr, Offset),
5987 MachinePointerInfo());
5988 }
5989
5990 // Return LR, which contains the return address. Mark it an implicit live-in.
5991 Register Reg = MF.addLiveIn(ARM::LR, getRegClassFor(MVT::i32));
5992 return DAG.getCopyFromReg(DAG.getEntryNode(), dl, Reg, VT);
5993}
5994
5995SDValue ARMTargetLowering::LowerFRAMEADDR(SDValue Op, SelectionDAG &DAG) const {
5996 const ARMBaseRegisterInfo &ARI =
5997 *static_cast<const ARMBaseRegisterInfo*>(RegInfo);
5999 MachineFrameInfo &MFI = MF.getFrameInfo();
6000 MFI.setFrameAddressIsTaken(true);
6001
6002 EVT VT = Op.getValueType();
6003 SDLoc dl(Op); // FIXME probably not meaningful
6004 unsigned Depth = Op.getConstantOperandVal(0);
6005 Register FrameReg = ARI.getFrameRegister(MF);
6006 SDValue FrameAddr = DAG.getCopyFromReg(DAG.getEntryNode(), dl, FrameReg, VT);
6007 while (Depth--)
6008 FrameAddr = DAG.getLoad(VT, dl, DAG.getEntryNode(), FrameAddr,
6009 MachinePointerInfo());
6010 return FrameAddr;
6011}
6012
6013// FIXME? Maybe this could be a TableGen attribute on some registers and
6014// this table could be generated automatically from RegInfo.
6015Register ARMTargetLowering::getRegisterByName(const char* RegName, LLT VT,
6016 const MachineFunction &MF) const {
6017 return StringSwitch<Register>(RegName)
6018 .Case("sp", ARM::SP)
6019 .Default(Register());
6020}
6021
6022// Result is 64 bit value so split into two 32 bit values and return as a
6023// pair of values.
6025 SelectionDAG &DAG) {
6026 SDLoc DL(N);
6027
6028 // This function is only supposed to be called for i64 type destination.
6029 assert(N->getValueType(0) == MVT::i64
6030 && "ExpandREAD_REGISTER called for non-i64 type result.");
6031
6033 DAG.getVTList(MVT::i32, MVT::i32, MVT::Other),
6034 N->getOperand(0),
6035 N->getOperand(1));
6036
6037 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Read.getValue(0),
6038 Read.getValue(1)));
6039 Results.push_back(Read.getValue(2)); // Chain
6040}
6041
6042/// \p BC is a bitcast that is about to be turned into a VMOVDRR.
6043/// When \p DstVT, the destination type of \p BC, is on the vector
6044/// register bank and the source of bitcast, \p Op, operates on the same bank,
6045/// it might be possible to combine them, such that everything stays on the
6046/// vector register bank.
6047/// \p return The node that would replace \p BT, if the combine
6048/// is possible.
6050 SelectionDAG &DAG) {
6051 SDValue Op = BC->getOperand(0);
6052 EVT DstVT = BC->getValueType(0);
6053
6054 // The only vector instruction that can produce a scalar (remember,
6055 // since the bitcast was about to be turned into VMOVDRR, the source
6056 // type is i64) from a vector is EXTRACT_VECTOR_ELT.
6057 // Moreover, we can do this combine only if there is one use.
6058 // Finally, if the destination type is not a vector, there is not
6059 // much point on forcing everything on the vector bank.
6060 if (!DstVT.isVector() || Op.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
6061 !Op.hasOneUse())
6062 return SDValue();
6063
6064 // If the index is not constant, we will introduce an additional
6065 // multiply that will stick.
6066 // Give up in that case.
6067 ConstantSDNode *Index = dyn_cast<ConstantSDNode>(Op.getOperand(1));
6068 if (!Index)
6069 return SDValue();
6070 unsigned DstNumElt = DstVT.getVectorNumElements();
6071
6072 // Compute the new index.
6073 const APInt &APIntIndex = Index->getAPIntValue();
6074 APInt NewIndex(APIntIndex.getBitWidth(), DstNumElt);
6075 NewIndex *= APIntIndex;
6076 // Check if the new constant index fits into i32.
6077 if (NewIndex.getBitWidth() > 32)
6078 return SDValue();
6079
6080 // vMTy bitcast(i64 extractelt vNi64 src, i32 index) ->
6081 // vMTy extractsubvector vNxMTy (bitcast vNi64 src), i32 index*M)
6082 SDLoc dl(Op);
6083 SDValue ExtractSrc = Op.getOperand(0);
6084 EVT VecVT = EVT::getVectorVT(
6085 *DAG.getContext(), DstVT.getScalarType(),
6086 ExtractSrc.getValueType().getVectorNumElements() * DstNumElt);
6087 SDValue BitCast = DAG.getNode(ISD::BITCAST, dl, VecVT, ExtractSrc);
6088 return DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DstVT, BitCast,
6089 DAG.getConstant(NewIndex.getZExtValue(), dl, MVT::i32));
6090}
6091
6092/// ExpandBITCAST - If the target supports VFP, this function is called to
6093/// expand a bit convert where either the source or destination type is i64 to
6094/// use a VMOVDRR or VMOVRRD node. This should not be done when the non-i64
6095/// operand type is illegal (e.g., v2f32 for a target that doesn't support
6096/// vectors), since the legalizer won't know what to do with that.
6097SDValue ARMTargetLowering::ExpandBITCAST(SDNode *N, SelectionDAG &DAG,
6098 const ARMSubtarget *Subtarget) const {
6099 SDLoc dl(N);
6100 SDValue Op = N->getOperand(0);
6101
6102 // This function is only supposed to be called for i16 and i64 types, either
6103 // as the source or destination of the bit convert.
6104 EVT SrcVT = Op.getValueType();
6105 EVT DstVT = N->getValueType(0);
6106
6107 if ((SrcVT == MVT::i16 || SrcVT == MVT::i32) &&
6108 (DstVT == MVT::f16 || DstVT == MVT::bf16))
6109 return MoveToHPR(SDLoc(N), DAG, MVT::i32, DstVT.getSimpleVT(),
6110 DAG.getNode(ISD::ZERO_EXTEND, SDLoc(N), MVT::i32, Op));
6111
6112 if ((DstVT == MVT::i16 || DstVT == MVT::i32) &&
6113 (SrcVT == MVT::f16 || SrcVT == MVT::bf16)) {
6114 if (Subtarget->hasFullFP16() && !Subtarget->hasBF16())
6115 Op = DAG.getBitcast(MVT::f16, Op);
6116 return DAG.getNode(
6117 ISD::TRUNCATE, SDLoc(N), DstVT,
6118 MoveFromHPR(SDLoc(N), DAG, MVT::i32, SrcVT.getSimpleVT(), Op));
6119 }
6120
6121 if (!(SrcVT == MVT::i64 || DstVT == MVT::i64))
6122 return SDValue();
6123
6124 // Turn i64->f64 into VMOVDRR.
6125 if (SrcVT == MVT::i64 && isTypeLegal(DstVT)) {
6126 // Do not force values to GPRs (this is what VMOVDRR does for the inputs)
6127 // if we can combine the bitcast with its source.
6128 if (SDValue Val = CombineVMOVDRRCandidateWithVecOp(N, DAG))
6129 return Val;
6130 SDValue Lo, Hi;
6131 std::tie(Lo, Hi) = DAG.SplitScalar(Op, dl, MVT::i32, MVT::i32);
6132 return DAG.getNode(ISD::BITCAST, dl, DstVT,
6133 DAG.getNode(ARMISD::VMOVDRR, dl, MVT::f64, Lo, Hi));
6134 }
6135
6136 // Turn f64->i64 into VMOVRRD.
6137 if (DstVT == MVT::i64 && isTypeLegal(SrcVT)) {
6138 SDValue Cvt;
6139 if (DAG.getDataLayout().isBigEndian() && SrcVT.isVector() &&
6140 SrcVT.getVectorNumElements() > 1)
6141 Cvt = DAG.getNode(ARMISD::VMOVRRD, dl,
6142 DAG.getVTList(MVT::i32, MVT::i32),
6143 DAG.getNode(ARMISD::VREV64, dl, SrcVT, Op));
6144 else
6145 Cvt = DAG.getNode(ARMISD::VMOVRRD, dl,
6146 DAG.getVTList(MVT::i32, MVT::i32), Op);
6147 // Merge the pieces into a single i64 value.
6148 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Cvt, Cvt.getValue(1));
6149 }
6150
6151 return SDValue();
6152}
6153
6154/// getZeroVector - Returns a vector of specified type with all zero elements.
6155/// Zero vectors are used to represent vector negation and in those cases
6156/// will be implemented with the NEON VNEG instruction. However, VNEG does
6157/// not support i64 elements, so sometimes the zero vectors will need to be
6158/// explicitly constructed. Regardless, use a canonical VMOV to create the
6159/// zero vector.
6160static SDValue getZeroVector(EVT VT, SelectionDAG &DAG, const SDLoc &dl) {
6161 assert(VT.isVector() && "Expected a vector type");
6162 // The canonical modified immediate encoding of a zero vector is....0!
6163 SDValue EncodedVal = DAG.getTargetConstant(0, dl, MVT::i32);
6164 EVT VmovVT = VT.is128BitVector() ? MVT::v4i32 : MVT::v2i32;
6165 SDValue Vmov = DAG.getNode(ARMISD::VMOVIMM, dl, VmovVT, EncodedVal);
6166 return DAG.getNode(ISD::BITCAST, dl, VT, Vmov);
6167}
6168
6169/// LowerShiftRightParts - Lower SRA_PARTS, which returns two
6170/// i32 values and take a 2 x i32 value to shift plus a shift amount.
6171SDValue ARMTargetLowering::LowerShiftRightParts(SDValue Op,
6172 SelectionDAG &DAG) const {
6173 assert(Op.getNumOperands() == 3 && "Not a double-shift!");
6174 EVT VT = Op.getValueType();
6175 unsigned VTBits = VT.getSizeInBits();
6176 SDLoc dl(Op);
6177 SDValue ShOpLo = Op.getOperand(0);
6178 SDValue ShOpHi = Op.getOperand(1);
6179 SDValue ShAmt = Op.getOperand(2);
6180 SDValue ARMcc;
6181 unsigned Opc = (Op.getOpcode() == ISD::SRA_PARTS) ? ISD::SRA : ISD::SRL;
6182
6183 assert(Op.getOpcode() == ISD::SRA_PARTS || Op.getOpcode() == ISD::SRL_PARTS);
6184
6185 SDValue RevShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32,
6186 DAG.getConstant(VTBits, dl, MVT::i32), ShAmt);
6187 SDValue Tmp1 = DAG.getNode(ISD::SRL, dl, VT, ShOpLo, ShAmt);
6188 SDValue ExtraShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, ShAmt,
6189 DAG.getConstant(VTBits, dl, MVT::i32));
6190 SDValue Tmp2 = DAG.getNode(ISD::SHL, dl, VT, ShOpHi, RevShAmt);
6191 SDValue LoSmallShift = DAG.getNode(ISD::OR, dl, VT, Tmp1, Tmp2);
6192 SDValue LoBigShift = DAG.getNode(Opc, dl, VT, ShOpHi, ExtraShAmt);
6193 SDValue CmpLo = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32),
6194 ISD::SETGE, ARMcc, DAG, dl);
6195 SDValue Lo =
6196 DAG.getNode(ARMISD::CMOV, dl, VT, LoSmallShift, LoBigShift, ARMcc, CmpLo);
6197
6198 SDValue HiSmallShift = DAG.getNode(Opc, dl, VT, ShOpHi, ShAmt);
6199 SDValue HiBigShift = Opc == ISD::SRA
6200 ? DAG.getNode(Opc, dl, VT, ShOpHi,
6201 DAG.getConstant(VTBits - 1, dl, VT))
6202 : DAG.getConstant(0, dl, VT);
6203 SDValue CmpHi = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32),
6204 ISD::SETGE, ARMcc, DAG, dl);
6205 SDValue Hi =
6206 DAG.getNode(ARMISD::CMOV, dl, VT, HiSmallShift, HiBigShift, ARMcc, CmpHi);
6207
6208 SDValue Ops[2] = { Lo, Hi };
6209 return DAG.getMergeValues(Ops, dl);
6210}
6211
6212/// LowerShiftLeftParts - Lower SHL_PARTS, which returns two
6213/// i32 values and take a 2 x i32 value to shift plus a shift amount.
6214SDValue ARMTargetLowering::LowerShiftLeftParts(SDValue Op,
6215 SelectionDAG &DAG) const {
6216 assert(Op.getNumOperands() == 3 && "Not a double-shift!");
6217 EVT VT = Op.getValueType();
6218 unsigned VTBits = VT.getSizeInBits();
6219 SDLoc dl(Op);
6220 SDValue ShOpLo = Op.getOperand(0);
6221 SDValue ShOpHi = Op.getOperand(1);
6222 SDValue ShAmt = Op.getOperand(2);
6223 SDValue ARMcc;
6224
6225 assert(Op.getOpcode() == ISD::SHL_PARTS);
6226 SDValue RevShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32,
6227 DAG.getConstant(VTBits, dl, MVT::i32), ShAmt);
6228 SDValue Tmp1 = DAG.getNode(ISD::SRL, dl, VT, ShOpLo, RevShAmt);
6229 SDValue Tmp2 = DAG.getNode(ISD::SHL, dl, VT, ShOpHi, ShAmt);
6230 SDValue HiSmallShift = DAG.getNode(ISD::OR, dl, VT, Tmp1, Tmp2);
6231
6232 SDValue ExtraShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32, ShAmt,
6233 DAG.getConstant(VTBits, dl, MVT::i32));
6234 SDValue HiBigShift = DAG.getNode(ISD::SHL, dl, VT, ShOpLo, ExtraShAmt);
6235 SDValue CmpHi = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32),
6236 ISD::SETGE, ARMcc, DAG, dl);
6237 SDValue Hi =
6238 DAG.getNode(ARMISD::CMOV, dl, VT, HiSmallShift, HiBigShift, ARMcc, CmpHi);
6239
6240 SDValue CmpLo = getARMCmp(ExtraShAmt, DAG.getConstant(0, dl, MVT::i32),
6241 ISD::SETGE, ARMcc, DAG, dl);
6242 SDValue LoSmallShift = DAG.getNode(ISD::SHL, dl, VT, ShOpLo, ShAmt);
6243 SDValue Lo = DAG.getNode(ARMISD::CMOV, dl, VT, LoSmallShift,
6244 DAG.getConstant(0, dl, VT), ARMcc, CmpLo);
6245
6246 SDValue Ops[2] = { Lo, Hi };
6247 return DAG.getMergeValues(Ops, dl);
6248}
6249
6250SDValue ARMTargetLowering::LowerGET_ROUNDING(SDValue Op,
6251 SelectionDAG &DAG) const {
6252 // The rounding mode is in bits 23:22 of the FPSCR.
6253 // The ARM rounding mode value to FLT_ROUNDS mapping is 0->1, 1->2, 2->3, 3->0
6254 // The formula we use to implement this is (((FPSCR + 1 << 22) >> 22) & 3)
6255 // so that the shift + and get folded into a bitfield extract.
6256 SDLoc dl(Op);
6257 SDValue Chain = Op.getOperand(0);
6258 SDValue Ops[] = {Chain,
6259 DAG.getConstant(Intrinsic::arm_get_fpscr, dl, MVT::i32)};
6260
6261 SDValue FPSCR =
6262 DAG.getNode(ISD::INTRINSIC_W_CHAIN, dl, {MVT::i32, MVT::Other}, Ops);
6263 Chain = FPSCR.getValue(1);
6264 SDValue FltRounds = DAG.getNode(ISD::ADD, dl, MVT::i32, FPSCR,
6265 DAG.getConstant(1U << 22, dl, MVT::i32));
6266 SDValue RMODE = DAG.getNode(ISD::SRL, dl, MVT::i32, FltRounds,
6267 DAG.getConstant(22, dl, MVT::i32));
6268 SDValue And = DAG.getNode(ISD::AND, dl, MVT::i32, RMODE,
6269 DAG.getConstant(3, dl, MVT::i32));
6270 return DAG.getMergeValues({And, Chain}, dl);
6271}
6272
6273SDValue ARMTargetLowering::LowerSET_ROUNDING(SDValue Op,
6274 SelectionDAG &DAG) const {
6275 SDLoc DL(Op);
6276 SDValue Chain = Op->getOperand(0);
6277 SDValue RMValue = Op->getOperand(1);
6278
6279 // The rounding mode is in bits 23:22 of the FPSCR.
6280 // The llvm.set.rounding argument value to ARM rounding mode value mapping
6281 // is 0->3, 1->0, 2->1, 3->2. The formula we use to implement this is
6282 // ((arg - 1) & 3) << 22).
6283 //
6284 // It is expected that the argument of llvm.set.rounding is within the
6285 // segment [0, 3], so NearestTiesToAway (4) is not handled here. It is
6286 // responsibility of the code generated llvm.set.rounding to ensure this
6287 // condition.
6288
6289 // Calculate new value of FPSCR[23:22].
6290 RMValue = DAG.getNode(ISD::SUB, DL, MVT::i32, RMValue,
6291 DAG.getConstant(1, DL, MVT::i32));
6292 RMValue = DAG.getNode(ISD::AND, DL, MVT::i32, RMValue,
6293 DAG.getConstant(0x3, DL, MVT::i32));
6294 RMValue = DAG.getNode(ISD::SHL, DL, MVT::i32, RMValue,
6295 DAG.getConstant(ARM::RoundingBitsPos, DL, MVT::i32));
6296
6297 // Get current value of FPSCR.
6298 SDValue Ops[] = {Chain,
6299 DAG.getConstant(Intrinsic::arm_get_fpscr, DL, MVT::i32)};
6300 SDValue FPSCR =
6301 DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, {MVT::i32, MVT::Other}, Ops);
6302 Chain = FPSCR.getValue(1);
6303 FPSCR = FPSCR.getValue(0);
6304
6305 // Put new rounding mode into FPSCR[23:22].
6306 const unsigned RMMask = ~(ARM::Rounding::rmMask << ARM::RoundingBitsPos);
6307 FPSCR = DAG.getNode(ISD::AND, DL, MVT::i32, FPSCR,
6308 DAG.getConstant(RMMask, DL, MVT::i32));
6309 FPSCR = DAG.getNode(ISD::OR, DL, MVT::i32, FPSCR, RMValue);
6310 SDValue Ops2[] = {
6311 Chain, DAG.getConstant(Intrinsic::arm_set_fpscr, DL, MVT::i32), FPSCR};
6312 return DAG.getNode(ISD::INTRINSIC_VOID, DL, MVT::Other, Ops2);
6313}
6314
6315SDValue ARMTargetLowering::LowerSET_FPMODE(SDValue Op,
6316 SelectionDAG &DAG) const {
6317 SDLoc DL(Op);
6318 SDValue Chain = Op->getOperand(0);
6319 SDValue Mode = Op->getOperand(1);
6320
6321 // Generate nodes to build:
6322 // FPSCR = (FPSCR & FPStatusBits) | (Mode & ~FPStatusBits)
6323 SDValue Ops[] = {Chain,
6324 DAG.getConstant(Intrinsic::arm_get_fpscr, DL, MVT::i32)};
6325 SDValue FPSCR =
6326 DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, {MVT::i32, MVT::Other}, Ops);
6327 Chain = FPSCR.getValue(1);
6328 FPSCR = FPSCR.getValue(0);
6329
6330 SDValue FPSCRMasked =
6331 DAG.getNode(ISD::AND, DL, MVT::i32, FPSCR,
6332 DAG.getConstant(ARM::FPStatusBits, DL, MVT::i32));
6333 SDValue InputMasked =
6334 DAG.getNode(ISD::AND, DL, MVT::i32, Mode,
6335 DAG.getConstant(~ARM::FPStatusBits, DL, MVT::i32));
6336 FPSCR = DAG.getNode(ISD::OR, DL, MVT::i32, FPSCRMasked, InputMasked);
6337
6338 SDValue Ops2[] = {
6339 Chain, DAG.getConstant(Intrinsic::arm_set_fpscr, DL, MVT::i32), FPSCR};
6340 return DAG.getNode(ISD::INTRINSIC_VOID, DL, MVT::Other, Ops2);
6341}
6342
6343SDValue ARMTargetLowering::LowerRESET_FPMODE(SDValue Op,
6344 SelectionDAG &DAG) const {
6345 SDLoc DL(Op);
6346 SDValue Chain = Op->getOperand(0);
6347
6348 // To get the default FP mode all control bits are cleared:
6349 // FPSCR = FPSCR & (FPStatusBits | FPReservedBits)
6350 SDValue Ops[] = {Chain,
6351 DAG.getConstant(Intrinsic::arm_get_fpscr, DL, MVT::i32)};
6352 SDValue FPSCR =
6353 DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, {MVT::i32, MVT::Other}, Ops);
6354 Chain = FPSCR.getValue(1);
6355 FPSCR = FPSCR.getValue(0);
6356
6357 SDValue FPSCRMasked = DAG.getNode(
6358 ISD::AND, DL, MVT::i32, FPSCR,
6360 SDValue Ops2[] = {Chain,
6361 DAG.getConstant(Intrinsic::arm_set_fpscr, DL, MVT::i32),
6362 FPSCRMasked};
6363 return DAG.getNode(ISD::INTRINSIC_VOID, DL, MVT::Other, Ops2);
6364}
6365
6367 const ARMSubtarget *ST) {
6368 SDLoc dl(N);
6369 EVT VT = N->getValueType(0);
6370 if (VT.isVector() && ST->hasNEON()) {
6371
6372 // Compute the least significant set bit: LSB = X & -X
6373 SDValue X = N->getOperand(0);
6374 SDValue NX = DAG.getNode(ISD::SUB, dl, VT, getZeroVector(VT, DAG, dl), X);
6375 SDValue LSB = DAG.getNode(ISD::AND, dl, VT, X, NX);
6376
6378
6379 if (ElemTy == MVT::i8) {
6380 // Compute with: cttz(x) = ctpop(lsb - 1)
6381 SDValue One = DAG.getNode(ARMISD::VMOVIMM, dl, VT,
6382 DAG.getTargetConstant(1, dl, ElemTy));
6383 SDValue Bits = DAG.getNode(ISD::SUB, dl, VT, LSB, One);
6384 return DAG.getNode(ISD::CTPOP, dl, VT, Bits);
6385 }
6386
6387 if ((ElemTy == MVT::i16 || ElemTy == MVT::i32) &&
6388 (N->getOpcode() == ISD::CTTZ_ZERO_POISON)) {
6389 // Compute with: cttz(x) = (width - 1) - ctlz(lsb), if x != 0
6390 unsigned NumBits = ElemTy.getSizeInBits();
6391 SDValue WidthMinus1 =
6392 DAG.getNode(ARMISD::VMOVIMM, dl, VT,
6393 DAG.getTargetConstant(NumBits - 1, dl, ElemTy));
6394 SDValue CTLZ = DAG.getNode(ISD::CTLZ, dl, VT, LSB);
6395 return DAG.getNode(ISD::SUB, dl, VT, WidthMinus1, CTLZ);
6396 }
6397
6398 // Compute with: cttz(x) = ctpop(lsb - 1)
6399
6400 // Compute LSB - 1.
6401 SDValue Bits;
6402 if (ElemTy == MVT::i64) {
6403 // Load constant 0xffff'ffff'ffff'ffff to register.
6404 SDValue FF = DAG.getNode(ARMISD::VMOVIMM, dl, VT,
6405 DAG.getTargetConstant(0x1eff, dl, MVT::i32));
6406 Bits = DAG.getNode(ISD::ADD, dl, VT, LSB, FF);
6407 } else {
6408 SDValue One = DAG.getNode(ARMISD::VMOVIMM, dl, VT,
6409 DAG.getTargetConstant(1, dl, ElemTy));
6410 Bits = DAG.getNode(ISD::SUB, dl, VT, LSB, One);
6411 }
6412 return DAG.getNode(ISD::CTPOP, dl, VT, Bits);
6413 }
6414
6415 if (!ST->hasV6T2Ops())
6416 return SDValue();
6417
6418 SDValue rbit = DAG.getNode(ISD::BITREVERSE, dl, VT, N->getOperand(0));
6419 return DAG.getNode(ISD::CTLZ, dl, VT, rbit);
6420}
6421
6423 const ARMSubtarget *ST) {
6424 EVT VT = N->getValueType(0);
6425 SDLoc DL(N);
6426
6427 assert(ST->hasNEON() && "Custom ctpop lowering requires NEON.");
6428 assert((VT == MVT::v1i64 || VT == MVT::v2i64 || VT == MVT::v2i32 ||
6429 VT == MVT::v4i32 || VT == MVT::v4i16 || VT == MVT::v8i16) &&
6430 "Unexpected type for custom ctpop lowering");
6431
6432 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
6433 EVT VT8Bit = VT.is64BitVector() ? MVT::v8i8 : MVT::v16i8;
6434 SDValue Res = DAG.getBitcast(VT8Bit, N->getOperand(0));
6435 Res = DAG.getNode(ISD::CTPOP, DL, VT8Bit, Res);
6436
6437 // Widen v8i8/v16i8 CTPOP result to VT by repeatedly widening pairwise adds.
6438 unsigned EltSize = 8;
6439 unsigned NumElts = VT.is64BitVector() ? 8 : 16;
6440 while (EltSize != VT.getScalarSizeInBits()) {
6442 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpaddlu, DL,
6443 TLI.getPointerTy(DAG.getDataLayout())));
6444 Ops.push_back(Res);
6445
6446 EltSize *= 2;
6447 NumElts /= 2;
6448 MVT WidenVT = MVT::getVectorVT(MVT::getIntegerVT(EltSize), NumElts);
6449 Res = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, WidenVT, Ops);
6450 }
6451
6452 return Res;
6453}
6454
6455/// Getvshiftimm - Check if this is a valid build_vector for the immediate
6456/// operand of a vector shift operation, where all the elements of the
6457/// build_vector must have the same constant integer value.
6458static bool getVShiftImm(SDValue Op, unsigned ElementBits, int64_t &Cnt) {
6459 // Ignore bit_converts.
6460 while (Op.getOpcode() == ISD::BITCAST)
6461 Op = Op.getOperand(0);
6463 APInt SplatBits, SplatUndef;
6464 unsigned SplatBitSize;
6465 bool HasAnyUndefs;
6466 if (!BVN ||
6467 !BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs,
6468 ElementBits) ||
6469 SplatBitSize > ElementBits)
6470 return false;
6471 Cnt = SplatBits.getSExtValue();
6472 return true;
6473}
6474
6475/// isVShiftLImm - Check if this is a valid build_vector for the immediate
6476/// operand of a vector shift left operation. That value must be in the range:
6477/// 0 <= Value < ElementBits for a left shift; or
6478/// 0 <= Value <= ElementBits for a long left shift.
6479static bool isVShiftLImm(SDValue Op, EVT VT, bool isLong, int64_t &Cnt) {
6480 assert(VT.isVector() && "vector shift count is not a vector type");
6481 int64_t ElementBits = VT.getScalarSizeInBits();
6482 if (!getVShiftImm(Op, ElementBits, Cnt))
6483 return false;
6484 return (Cnt >= 0 && (isLong ? Cnt - 1 : Cnt) < ElementBits);
6485}
6486
6487/// isVShiftRImm - Check if this is a valid build_vector for the immediate
6488/// operand of a vector shift right operation. For a shift opcode, the value
6489/// is positive, but for an intrinsic the value count must be negative. The
6490/// absolute value must be in the range:
6491/// 1 <= |Value| <= ElementBits for a right shift; or
6492/// 1 <= |Value| <= ElementBits/2 for a narrow right shift.
6493static bool isVShiftRImm(SDValue Op, EVT VT, bool isNarrow, bool isIntrinsic,
6494 int64_t &Cnt) {
6495 assert(VT.isVector() && "vector shift count is not a vector type");
6496 int64_t ElementBits = VT.getScalarSizeInBits();
6497 if (!getVShiftImm(Op, ElementBits, Cnt))
6498 return false;
6499 if (!isIntrinsic)
6500 return (Cnt >= 1 && Cnt <= (isNarrow ? ElementBits / 2 : ElementBits));
6501 if (Cnt >= -(isNarrow ? ElementBits / 2 : ElementBits) && Cnt <= -1) {
6502 Cnt = -Cnt;
6503 return true;
6504 }
6505 return false;
6506}
6507
6509 const ARMSubtarget *ST) {
6510 EVT VT = N->getValueType(0);
6511 SDLoc dl(N);
6512 int64_t Cnt;
6513
6514 if (!VT.isVector())
6515 return SDValue();
6516
6517 // We essentially have two forms here. Shift by an immediate and shift by a
6518 // vector register (there are also shift by a gpr, but that is just handled
6519 // with a tablegen pattern). We cannot easily match shift by an immediate in
6520 // tablegen so we do that here and generate a VSHLIMM/VSHRsIMM/VSHRuIMM.
6521 // For shifting by a vector, we don't have VSHR, only VSHL (which can be
6522 // signed or unsigned, and a negative shift indicates a shift right).
6523 if (N->getOpcode() == ISD::SHL) {
6524 if (isVShiftLImm(N->getOperand(1), VT, false, Cnt))
6525 return DAG.getNode(ARMISD::VSHLIMM, dl, VT, N->getOperand(0),
6526 DAG.getConstant(Cnt, dl, MVT::i32));
6527 return DAG.getNode(ARMISD::VSHLu, dl, VT, N->getOperand(0),
6528 N->getOperand(1));
6529 }
6530
6531 assert((N->getOpcode() == ISD::SRA || N->getOpcode() == ISD::SRL) &&
6532 "unexpected vector shift opcode");
6533
6534 if (isVShiftRImm(N->getOperand(1), VT, false, false, Cnt)) {
6535 unsigned VShiftOpc =
6536 (N->getOpcode() == ISD::SRA ? ARMISD::VSHRsIMM : ARMISD::VSHRuIMM);
6537 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0),
6538 DAG.getConstant(Cnt, dl, MVT::i32));
6539 }
6540
6541 // Other right shifts we don't have operations for (we use a shift left by a
6542 // negative number).
6543 EVT ShiftVT = N->getOperand(1).getValueType();
6544 SDValue NegatedCount = DAG.getNode(
6545 ISD::SUB, dl, ShiftVT, getZeroVector(ShiftVT, DAG, dl), N->getOperand(1));
6546 unsigned VShiftOpc =
6547 (N->getOpcode() == ISD::SRA ? ARMISD::VSHLs : ARMISD::VSHLu);
6548 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0), NegatedCount);
6549}
6550
6552 const ARMSubtarget *ST) {
6553 EVT VT = N->getValueType(0);
6554 SDLoc dl(N);
6555
6556 // We can get here for a node like i32 = ISD::SHL i32, i64
6557 if (VT != MVT::i64)
6558 return SDValue();
6559
6560 assert((N->getOpcode() == ISD::SRL || N->getOpcode() == ISD::SRA ||
6561 N->getOpcode() == ISD::SHL) &&
6562 "Unknown shift to lower!");
6563
6564 unsigned ShOpc = N->getOpcode();
6565 if (ST->hasMVEIntegerOps()) {
6566 SDValue ShAmt = N->getOperand(1);
6567 unsigned ShPartsOpc = ARMISD::LSLL;
6569
6570 // If the shift amount is greater than 32 or has a greater bitwidth than 64
6571 // then do the default optimisation
6572 if ((!Con && ShAmt->getValueType(0).getSizeInBits() > 64) ||
6573 (Con && (Con->getAPIntValue() == 0 || Con->getAPIntValue().uge(32))))
6574 return SDValue();
6575
6576 // Extract the lower 32 bits of the shift amount if it's not an i32
6577 if (ShAmt->getValueType(0) != MVT::i32)
6578 ShAmt = DAG.getZExtOrTrunc(ShAmt, dl, MVT::i32);
6579
6580 if (ShOpc == ISD::SRL) {
6581 if (!Con)
6582 // There is no t2LSRLr instruction so negate and perform an lsll if the
6583 // shift amount is in a register, emulating a right shift.
6584 ShAmt = DAG.getNode(ISD::SUB, dl, MVT::i32,
6585 DAG.getConstant(0, dl, MVT::i32), ShAmt);
6586 else
6587 // Else generate an lsrl on the immediate shift amount
6588 ShPartsOpc = ARMISD::LSRL;
6589 } else if (ShOpc == ISD::SRA)
6590 ShPartsOpc = ARMISD::ASRL;
6591
6592 // Split Lower/Upper 32 bits of the destination/source
6593 SDValue Lo, Hi;
6594 std::tie(Lo, Hi) =
6595 DAG.SplitScalar(N->getOperand(0), dl, MVT::i32, MVT::i32);
6596 // Generate the shift operation as computed above
6597 Lo = DAG.getNode(ShPartsOpc, dl, DAG.getVTList(MVT::i32, MVT::i32), Lo, Hi,
6598 ShAmt);
6599 // The upper 32 bits come from the second return value of lsll
6600 Hi = SDValue(Lo.getNode(), 1);
6601 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lo, Hi);
6602 }
6603
6604 // We only lower SRA, SRL of 1 here, all others use generic lowering.
6605 if (!isOneConstant(N->getOperand(1)) || N->getOpcode() == ISD::SHL)
6606 return SDValue();
6607
6608 // If we are in thumb mode, we don't have RRX.
6609 if (ST->isThumb1Only())
6610 return SDValue();
6611
6612 // Okay, we have a 64-bit SRA or SRL of 1. Lower this to an RRX expr.
6613 SDValue Lo, Hi;
6614 std::tie(Lo, Hi) = DAG.SplitScalar(N->getOperand(0), dl, MVT::i32, MVT::i32);
6615
6616 // First, build a LSRS1/ASRS1 op, which shifts the top part by one and
6617 // captures the shifted out bit into a carry flag.
6618 unsigned Opc = N->getOpcode() == ISD::SRL ? ARMISD::LSRS1 : ARMISD::ASRS1;
6619 Hi = DAG.getNode(Opc, dl, DAG.getVTList(MVT::i32, FlagsVT), Hi);
6620
6621 // The low part is an ARMISD::RRX operand, which shifts the carry in.
6622 Lo = DAG.getNode(ARMISD::RRX, dl, MVT::i32, Lo, Hi.getValue(1));
6623
6624 // Merge the pieces into a single i64 value.
6625 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lo, Hi);
6626}
6627
6629 const ARMSubtarget *ST) {
6630 bool Invert = false;
6631 bool Swap = false;
6632 unsigned Opc = ARMCC::AL;
6633
6634 SDValue Op0 = Op.getOperand(0);
6635 SDValue Op1 = Op.getOperand(1);
6636 SDValue CC = Op.getOperand(2);
6637 EVT VT = Op.getValueType();
6638 ISD::CondCode SetCCOpcode = cast<CondCodeSDNode>(CC)->get();
6639 SDLoc dl(Op);
6640
6641 EVT CmpVT;
6642 if (ST->hasNEON())
6644 else {
6645 assert(ST->hasMVEIntegerOps() &&
6646 "No hardware support for integer vector comparison!");
6647
6648 if (Op.getValueType().getVectorElementType() != MVT::i1)
6649 return SDValue();
6650
6651 // Make sure we expand floating point setcc to scalar if we do not have
6652 // mve.fp, so that we can handle them from there.
6653 if (Op0.getValueType().isFloatingPoint() && !ST->hasMVEFloatOps())
6654 return SDValue();
6655
6656 CmpVT = VT;
6657 }
6658
6659 if (Op0.getValueType().getVectorElementType() == MVT::i64 &&
6660 (SetCCOpcode == ISD::SETEQ || SetCCOpcode == ISD::SETNE)) {
6661 // Special-case integer 64-bit equality comparisons. They aren't legal,
6662 // but they can be lowered with a few vector instructions.
6663 unsigned CmpElements = CmpVT.getVectorNumElements() * 2;
6664 EVT SplitVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, CmpElements);
6665 SDValue CastOp0 = DAG.getNode(ISD::BITCAST, dl, SplitVT, Op0);
6666 SDValue CastOp1 = DAG.getNode(ISD::BITCAST, dl, SplitVT, Op1);
6667 SDValue Cmp = DAG.getNode(ISD::SETCC, dl, SplitVT, CastOp0, CastOp1,
6668 DAG.getCondCode(ISD::SETEQ));
6669 SDValue Reversed = DAG.getNode(ARMISD::VREV64, dl, SplitVT, Cmp);
6670 SDValue Merged = DAG.getNode(ISD::AND, dl, SplitVT, Cmp, Reversed);
6671 Merged = DAG.getNode(ISD::BITCAST, dl, CmpVT, Merged);
6672 if (SetCCOpcode == ISD::SETNE)
6673 Merged = DAG.getNOT(dl, Merged, CmpVT);
6674 Merged = DAG.getSExtOrTrunc(Merged, dl, VT);
6675 return Merged;
6676 }
6677
6678 if (CmpVT.getVectorElementType() == MVT::i64)
6679 // 64-bit comparisons are not legal in general.
6680 return SDValue();
6681
6682 if (Op1.getValueType().isFloatingPoint()) {
6683 switch (SetCCOpcode) {
6684 default: llvm_unreachable("Illegal FP comparison");
6685 case ISD::SETUNE:
6686 case ISD::SETNE:
6687 if (ST->hasMVEFloatOps()) {
6688 Opc = ARMCC::NE; break;
6689 } else {
6690 Invert = true; [[fallthrough]];
6691 }
6692 case ISD::SETOEQ:
6693 case ISD::SETEQ: Opc = ARMCC::EQ; break;
6694 case ISD::SETOLT:
6695 case ISD::SETLT: Swap = true; [[fallthrough]];
6696 case ISD::SETOGT:
6697 case ISD::SETGT: Opc = ARMCC::GT; break;
6698 case ISD::SETOLE:
6699 case ISD::SETLE: Swap = true; [[fallthrough]];
6700 case ISD::SETOGE:
6701 case ISD::SETGE: Opc = ARMCC::GE; break;
6702 case ISD::SETUGE: Swap = true; [[fallthrough]];
6703 case ISD::SETULE: Invert = true; Opc = ARMCC::GT; break;
6704 case ISD::SETUGT: Swap = true; [[fallthrough]];
6705 case ISD::SETULT: Invert = true; Opc = ARMCC::GE; break;
6706 case ISD::SETUEQ: Invert = true; [[fallthrough]];
6707 case ISD::SETONE: {
6708 // Expand this to (OLT | OGT).
6709 SDValue TmpOp0 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op1, Op0,
6710 DAG.getConstant(ARMCC::GT, dl, MVT::i32));
6711 SDValue TmpOp1 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1,
6712 DAG.getConstant(ARMCC::GT, dl, MVT::i32));
6713 SDValue Result = DAG.getNode(ISD::OR, dl, CmpVT, TmpOp0, TmpOp1);
6714 if (Invert)
6715 Result = DAG.getNOT(dl, Result, VT);
6716 return Result;
6717 }
6718 case ISD::SETUO: Invert = true; [[fallthrough]];
6719 case ISD::SETO: {
6720 // Expand this to (OLT | OGE).
6721 SDValue TmpOp0 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op1, Op0,
6722 DAG.getConstant(ARMCC::GT, dl, MVT::i32));
6723 SDValue TmpOp1 = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1,
6724 DAG.getConstant(ARMCC::GE, dl, MVT::i32));
6725 SDValue Result = DAG.getNode(ISD::OR, dl, CmpVT, TmpOp0, TmpOp1);
6726 if (Invert)
6727 Result = DAG.getNOT(dl, Result, VT);
6728 return Result;
6729 }
6730 }
6731 } else {
6732 // Integer comparisons.
6733 switch (SetCCOpcode) {
6734 default: llvm_unreachable("Illegal integer comparison");
6735 case ISD::SETNE:
6736 if (ST->hasMVEIntegerOps()) {
6737 Opc = ARMCC::NE; break;
6738 } else {
6739 Invert = true; [[fallthrough]];
6740 }
6741 case ISD::SETEQ: Opc = ARMCC::EQ; break;
6742 case ISD::SETLT: Swap = true; [[fallthrough]];
6743 case ISD::SETGT: Opc = ARMCC::GT; break;
6744 case ISD::SETLE: Swap = true; [[fallthrough]];
6745 case ISD::SETGE: Opc = ARMCC::GE; break;
6746 case ISD::SETULT: Swap = true; [[fallthrough]];
6747 case ISD::SETUGT: Opc = ARMCC::HI; break;
6748 case ISD::SETULE: Swap = true; [[fallthrough]];
6749 case ISD::SETUGE: Opc = ARMCC::HS; break;
6750 }
6751
6752 // Detect VTST (Vector Test Bits) = icmp ne (and (op0, op1), zero).
6753 if (ST->hasNEON() && Opc == ARMCC::EQ) {
6754 SDValue AndOp;
6756 AndOp = Op0;
6757 else if (ISD::isBuildVectorAllZeros(Op0.getNode()))
6758 AndOp = Op1;
6759
6760 // Ignore bitconvert.
6761 if (AndOp.getNode() && AndOp.getOpcode() == ISD::BITCAST)
6762 AndOp = AndOp.getOperand(0);
6763
6764 if (AndOp.getNode() && AndOp.getOpcode() == ISD::AND) {
6765 Op0 = DAG.getNode(ISD::BITCAST, dl, CmpVT, AndOp.getOperand(0));
6766 Op1 = DAG.getNode(ISD::BITCAST, dl, CmpVT, AndOp.getOperand(1));
6767 SDValue Result = DAG.getNode(ARMISD::VTST, dl, CmpVT, Op0, Op1);
6768 if (!Invert)
6769 Result = DAG.getNOT(dl, Result, VT);
6770 return Result;
6771 }
6772 }
6773 }
6774
6775 if (Swap)
6776 std::swap(Op0, Op1);
6777
6778 // If one of the operands is a constant vector zero, attempt to fold the
6779 // comparison to a specialized compare-against-zero form.
6781 (Opc == ARMCC::GE || Opc == ARMCC::GT || Opc == ARMCC::EQ ||
6782 Opc == ARMCC::NE)) {
6783 if (Opc == ARMCC::GE)
6784 Opc = ARMCC::LE;
6785 else if (Opc == ARMCC::GT)
6786 Opc = ARMCC::LT;
6787 std::swap(Op0, Op1);
6788 }
6789
6790 SDValue Result;
6792 (Opc == ARMCC::GE || Opc == ARMCC::GT || Opc == ARMCC::LE ||
6793 Opc == ARMCC::LT || Opc == ARMCC::NE || Opc == ARMCC::EQ))
6794 Result = DAG.getNode(ARMISD::VCMPZ, dl, CmpVT, Op0,
6795 DAG.getConstant(Opc, dl, MVT::i32));
6796 else
6797 Result = DAG.getNode(ARMISD::VCMP, dl, CmpVT, Op0, Op1,
6798 DAG.getConstant(Opc, dl, MVT::i32));
6799
6800 Result = DAG.getSExtOrTrunc(Result, dl, VT);
6801
6802 if (Invert)
6803 Result = DAG.getNOT(dl, Result, VT);
6804
6805 return Result;
6806}
6807
6809 SDValue LHS = Op.getOperand(0);
6810 SDValue RHS = Op.getOperand(1);
6811
6812 assert(LHS.getSimpleValueType().isInteger() && "SETCCCARRY is integer only.");
6813
6814 SDValue Carry = Op.getOperand(2);
6815 SDValue Cond = Op.getOperand(3);
6816 SDLoc DL(Op);
6817
6818 // ARMISD::SUBE expects a carry not a borrow like ISD::USUBO_CARRY so we
6819 // have to invert the carry first.
6820 SDValue InvCarry = valueToCarryFlag(Carry, DAG, true);
6821
6822 SDVTList VTs = DAG.getVTList(LHS.getValueType(), MVT::i32);
6823 SDValue Cmp = DAG.getNode(ARMISD::SUBE, DL, VTs, LHS, RHS, InvCarry);
6824
6825 SDValue FVal = DAG.getConstant(0, DL, MVT::i32);
6826 SDValue TVal = DAG.getConstant(1, DL, MVT::i32);
6827 SDValue ARMcc = DAG.getConstant(
6828 IntCCToARMCC(cast<CondCodeSDNode>(Cond)->get()), DL, MVT::i32);
6829 return DAG.getNode(ARMISD::CMOV, DL, Op.getValueType(), FVal, TVal, ARMcc,
6830 Cmp.getValue(1));
6831}
6832
6833/// isVMOVModifiedImm - Check if the specified splat value corresponds to a
6834/// valid vector constant for a NEON or MVE instruction with a "modified
6835/// immediate" operand (e.g., VMOV). If so, return the encoded value.
6836static SDValue isVMOVModifiedImm(uint64_t SplatBits, uint64_t SplatUndef,
6837 unsigned SplatBitSize, SelectionDAG &DAG,
6838 const SDLoc &dl, EVT &VT, EVT VectorVT,
6839 VMOVModImmType type) {
6840 unsigned OpCmode, Imm;
6841 bool is128Bits = VectorVT.is128BitVector();
6842
6843 // SplatBitSize is set to the smallest size that splats the vector, so a
6844 // zero vector will always have SplatBitSize == 8. However, NEON modified
6845 // immediate instructions others than VMOV do not support the 8-bit encoding
6846 // of a zero vector, and the default encoding of zero is supposed to be the
6847 // 32-bit version.
6848 if (SplatBits == 0)
6849 SplatBitSize = 32;
6850
6851 switch (SplatBitSize) {
6852 case 8:
6853 if (type != VMOVModImm)
6854 return SDValue();
6855 // Any 1-byte value is OK. Op=0, Cmode=1110.
6856 assert((SplatBits & ~0xff) == 0 && "one byte splat value is too big");
6857 OpCmode = 0xe;
6858 Imm = SplatBits;
6859 VT = is128Bits ? MVT::v16i8 : MVT::v8i8;
6860 break;
6861
6862 case 16:
6863 // NEON's 16-bit VMOV supports splat values where only one byte is nonzero.
6864 VT = is128Bits ? MVT::v8i16 : MVT::v4i16;
6865 if ((SplatBits & ~0xff) == 0) {
6866 // Value = 0x00nn: Op=x, Cmode=100x.
6867 OpCmode = 0x8;
6868 Imm = SplatBits;
6869 break;
6870 }
6871 if ((SplatBits & ~0xff00) == 0) {
6872 // Value = 0xnn00: Op=x, Cmode=101x.
6873 OpCmode = 0xa;
6874 Imm = SplatBits >> 8;
6875 break;
6876 }
6877 return SDValue();
6878
6879 case 32:
6880 // NEON's 32-bit VMOV supports splat values where:
6881 // * only one byte is nonzero, or
6882 // * the least significant byte is 0xff and the second byte is nonzero, or
6883 // * the least significant 2 bytes are 0xff and the third is nonzero.
6884 VT = is128Bits ? MVT::v4i32 : MVT::v2i32;
6885 if ((SplatBits & ~0xff) == 0) {
6886 // Value = 0x000000nn: Op=x, Cmode=000x.
6887 OpCmode = 0;
6888 Imm = SplatBits;
6889 break;
6890 }
6891 if ((SplatBits & ~0xff00) == 0) {
6892 // Value = 0x0000nn00: Op=x, Cmode=001x.
6893 OpCmode = 0x2;
6894 Imm = SplatBits >> 8;
6895 break;
6896 }
6897 if ((SplatBits & ~0xff0000) == 0) {
6898 // Value = 0x00nn0000: Op=x, Cmode=010x.
6899 OpCmode = 0x4;
6900 Imm = SplatBits >> 16;
6901 break;
6902 }
6903 if ((SplatBits & ~0xff000000) == 0) {
6904 // Value = 0xnn000000: Op=x, Cmode=011x.
6905 OpCmode = 0x6;
6906 Imm = SplatBits >> 24;
6907 break;
6908 }
6909
6910 // cmode == 0b1100 and cmode == 0b1101 are not supported for VORR or VBIC
6911 if (type == OtherModImm) return SDValue();
6912
6913 if ((SplatBits & ~0xffff) == 0 &&
6914 ((SplatBits | SplatUndef) & 0xff) == 0xff) {
6915 // Value = 0x0000nnff: Op=x, Cmode=1100.
6916 OpCmode = 0xc;
6917 Imm = SplatBits >> 8;
6918 break;
6919 }
6920
6921 // cmode == 0b1101 is not supported for MVE VMVN
6922 if (type == MVEVMVNModImm)
6923 return SDValue();
6924
6925 if ((SplatBits & ~0xffffff) == 0 &&
6926 ((SplatBits | SplatUndef) & 0xffff) == 0xffff) {
6927 // Value = 0x00nnffff: Op=x, Cmode=1101.
6928 OpCmode = 0xd;
6929 Imm = SplatBits >> 16;
6930 break;
6931 }
6932
6933 // Note: there are a few 32-bit splat values (specifically: 00ffff00,
6934 // ff000000, ff0000ff, and ffff00ff) that are valid for VMOV.I64 but not
6935 // VMOV.I32. A (very) minor optimization would be to replicate the value
6936 // and fall through here to test for a valid 64-bit splat. But, then the
6937 // caller would also need to check and handle the change in size.
6938 return SDValue();
6939
6940 case 64: {
6941 if (type != VMOVModImm)
6942 return SDValue();
6943 // NEON has a 64-bit VMOV splat where each byte is either 0 or 0xff.
6944 uint64_t BitMask = 0xff;
6945 unsigned ImmMask = 1;
6946 Imm = 0;
6947 for (int ByteNum = 0; ByteNum < 8; ++ByteNum) {
6948 if (((SplatBits | SplatUndef) & BitMask) == BitMask) {
6949 Imm |= ImmMask;
6950 } else if ((SplatBits & BitMask) != 0) {
6951 return SDValue();
6952 }
6953 BitMask <<= 8;
6954 ImmMask <<= 1;
6955 }
6956
6957 // Op=1, Cmode=1110.
6958 OpCmode = 0x1e;
6959 VT = is128Bits ? MVT::v2i64 : MVT::v1i64;
6960 break;
6961 }
6962
6963 default:
6964 llvm_unreachable("unexpected size for isVMOVModifiedImm");
6965 }
6966
6967 unsigned EncodedVal = ARM_AM::createVMOVModImm(OpCmode, Imm);
6968 return DAG.getTargetConstant(EncodedVal, dl, MVT::i32);
6969}
6970
6971SDValue ARMTargetLowering::LowerConstantFP(SDValue Op, SelectionDAG &DAG,
6972 const ARMSubtarget *ST) const {
6973 EVT VT = Op.getValueType();
6974 bool IsDouble = (VT == MVT::f64);
6975 ConstantFPSDNode *CFP = cast<ConstantFPSDNode>(Op);
6976 const APFloat &FPVal = CFP->getValueAPF();
6977
6978 // Prevent floating-point constants from using literal loads
6979 // when execute-only is enabled.
6980 if (ST->genExecuteOnly()) {
6981 // We shouldn't trigger this for v6m execute-only
6982 assert((!ST->isThumb1Only() || ST->hasV8MBaselineOps()) &&
6983 "Unexpected architecture");
6984
6985 // If we can represent the constant as an immediate, don't lower it
6986 if (isFPImmLegal(FPVal, VT))
6987 return Op;
6988 // Otherwise, construct as integer, and move to float register
6989 APInt INTVal = FPVal.bitcastToAPInt();
6990 SDLoc DL(CFP);
6991 switch (VT.getSimpleVT().SimpleTy) {
6992 default:
6993 llvm_unreachable("Unknown floating point type!");
6994 break;
6995 case MVT::f64: {
6996 SDValue Lo = DAG.getConstant(INTVal.trunc(32), DL, MVT::i32);
6997 SDValue Hi = DAG.getConstant(INTVal.lshr(32).trunc(32), DL, MVT::i32);
6998 return DAG.getNode(ARMISD::VMOVDRR, DL, MVT::f64, Lo, Hi);
6999 }
7000 case MVT::f32:
7001 return DAG.getNode(ARMISD::VMOVSR, DL, VT,
7002 DAG.getConstant(INTVal, DL, MVT::i32));
7003 }
7004 }
7005
7006 if (!ST->hasVFP3Base())
7007 return SDValue();
7008
7009 // Use the default (constant pool) lowering for double constants when we have
7010 // an SP-only FPU
7011 if (IsDouble && !Subtarget->hasFP64())
7012 return SDValue();
7013
7014 // Try splatting with a VMOV.f32...
7015 int ImmVal = IsDouble ? ARM_AM::getFP64Imm(FPVal) : ARM_AM::getFP32Imm(FPVal);
7016
7017 if (ImmVal != -1) {
7018 if (IsDouble || !ST->useNEONForSinglePrecisionFP()) {
7019 // We have code in place to select a valid ConstantFP already, no need to
7020 // do any mangling.
7021 return Op;
7022 }
7023
7024 // It's a float and we are trying to use NEON operations where
7025 // possible. Lower it to a splat followed by an extract.
7026 SDLoc DL(Op);
7027 SDValue NewVal = DAG.getTargetConstant(ImmVal, DL, MVT::i32);
7028 SDValue VecConstant = DAG.getNode(ARMISD::VMOVFPIMM, DL, MVT::v2f32,
7029 NewVal);
7030 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecConstant,
7031 DAG.getConstant(0, DL, MVT::i32));
7032 }
7033
7034 // The rest of our options are NEON only, make sure that's allowed before
7035 // proceeding..
7036 if (!ST->hasNEON() || (!IsDouble && !ST->useNEONForSinglePrecisionFP()))
7037 return SDValue();
7038
7039 EVT VMovVT;
7040 uint64_t iVal = FPVal.bitcastToAPInt().getZExtValue();
7041
7042 // It wouldn't really be worth bothering for doubles except for one very
7043 // important value, which does happen to match: 0.0. So make sure we don't do
7044 // anything stupid.
7045 if (IsDouble && (iVal & 0xffffffff) != (iVal >> 32))
7046 return SDValue();
7047
7048 // Try a VMOV.i32 (FIXME: i8, i16, or i64 could work too).
7049 SDValue NewVal = isVMOVModifiedImm(iVal & 0xffffffffU, 0, 32, DAG, SDLoc(Op),
7050 VMovVT, VT, VMOVModImm);
7051 if (NewVal != SDValue()) {
7052 SDLoc DL(Op);
7053 SDValue VecConstant = DAG.getNode(ARMISD::VMOVIMM, DL, VMovVT,
7054 NewVal);
7055 if (IsDouble)
7056 return DAG.getNode(ISD::BITCAST, DL, MVT::f64, VecConstant);
7057
7058 // It's a float: cast and extract a vector element.
7059 SDValue VecFConstant = DAG.getNode(ISD::BITCAST, DL, MVT::v2f32,
7060 VecConstant);
7061 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecFConstant,
7062 DAG.getConstant(0, DL, MVT::i32));
7063 }
7064
7065 // Finally, try a VMVN.i32
7066 NewVal = isVMOVModifiedImm(~iVal & 0xffffffffU, 0, 32, DAG, SDLoc(Op), VMovVT,
7067 VT, VMVNModImm);
7068 if (NewVal != SDValue()) {
7069 SDLoc DL(Op);
7070 SDValue VecConstant = DAG.getNode(ARMISD::VMVNIMM, DL, VMovVT, NewVal);
7071
7072 if (IsDouble)
7073 return DAG.getNode(ISD::BITCAST, DL, MVT::f64, VecConstant);
7074
7075 // It's a float: cast and extract a vector element.
7076 SDValue VecFConstant = DAG.getNode(ISD::BITCAST, DL, MVT::v2f32,
7077 VecConstant);
7078 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, VecFConstant,
7079 DAG.getConstant(0, DL, MVT::i32));
7080 }
7081
7082 return SDValue();
7083}
7084
7085// check if an VEXT instruction can handle the shuffle mask when the
7086// vector sources of the shuffle are the same.
7087static bool isSingletonVEXTMask(ArrayRef<int> M, EVT VT, unsigned &Imm) {
7088 unsigned NumElts = VT.getVectorNumElements();
7089
7090 // Assume that the first shuffle index is not UNDEF. Fail if it is.
7091 if (M[0] < 0)
7092 return false;
7093
7094 Imm = M[0];
7095
7096 // If this is a VEXT shuffle, the immediate value is the index of the first
7097 // element. The other shuffle indices must be the successive elements after
7098 // the first one.
7099 unsigned ExpectedElt = Imm;
7100 for (unsigned i = 1; i < NumElts; ++i) {
7101 // Increment the expected index. If it wraps around, just follow it
7102 // back to index zero and keep going.
7103 ++ExpectedElt;
7104 if (ExpectedElt == NumElts)
7105 ExpectedElt = 0;
7106
7107 if (M[i] < 0) continue; // ignore UNDEF indices
7108 if (ExpectedElt != static_cast<unsigned>(M[i]))
7109 return false;
7110 }
7111
7112 return true;
7113}
7114
7115static bool isVEXTMask(ArrayRef<int> M, EVT VT,
7116 bool &ReverseVEXT, unsigned &Imm) {
7117 unsigned NumElts = VT.getVectorNumElements();
7118 ReverseVEXT = false;
7119
7120 // Assume that the first shuffle index is not UNDEF. Fail if it is.
7121 if (M[0] < 0)
7122 return false;
7123
7124 Imm = M[0];
7125
7126 // If this is a VEXT shuffle, the immediate value is the index of the first
7127 // element. The other shuffle indices must be the successive elements after
7128 // the first one.
7129 unsigned ExpectedElt = Imm;
7130 for (unsigned i = 1; i < NumElts; ++i) {
7131 // Increment the expected index. If it wraps around, it may still be
7132 // a VEXT but the source vectors must be swapped.
7133 ExpectedElt += 1;
7134 if (ExpectedElt == NumElts * 2) {
7135 ExpectedElt = 0;
7136 ReverseVEXT = true;
7137 }
7138
7139 if (M[i] < 0) continue; // ignore UNDEF indices
7140 if (ExpectedElt != static_cast<unsigned>(M[i]))
7141 return false;
7142 }
7143
7144 // Adjust the index value if the source operands will be swapped.
7145 if (ReverseVEXT)
7146 Imm -= NumElts;
7147
7148 return true;
7149}
7150
7151static bool isVTBLMask(ArrayRef<int> M, EVT VT) {
7152 // We can handle <8 x i8> vector shuffles. If the index in the mask is out of
7153 // range, then 0 is placed into the resulting vector. So pretty much any mask
7154 // of 8 elements can work here.
7155 return VT == MVT::v8i8 && M.size() == 8;
7156}
7157
7158/// Check if \p ShuffleMask is a NEON two-result shuffle (VZIP, VUZP, VTRN),
7159/// and return the corresponding ARMISD opcode if it is, or 0 if it isn't.
7160static unsigned isNEONTwoResultShuffleMask(ArrayRef<int> ShuffleMask, EVT VT,
7161 unsigned &WhichResult,
7162 bool &isV_UNDEF) {
7163 isV_UNDEF = false;
7164 if (isVTRNMask(ShuffleMask, VT, WhichResult))
7165 return ARMISD::VTRN;
7166 if (isVUZPMask(ShuffleMask, VT, WhichResult))
7167 return ARMISD::VUZP;
7168 if (isVZIPMask(ShuffleMask, VT, WhichResult))
7169 return ARMISD::VZIP;
7170
7171 isV_UNDEF = true;
7172 if (isVTRN_v_undef_Mask(ShuffleMask, VT, WhichResult))
7173 return ARMISD::VTRN;
7174 if (isVUZP_v_undef_Mask(ShuffleMask, VT, WhichResult))
7175 return ARMISD::VUZP;
7176 if (isVZIP_v_undef_Mask(ShuffleMask, VT, WhichResult))
7177 return ARMISD::VZIP;
7178
7179 return 0;
7180}
7181
7182/// \return true if this is a reverse operation on an vector.
7183static bool isReverseMask(ArrayRef<int> M, EVT VT) {
7184 unsigned NumElts = VT.getVectorNumElements();
7185 // Make sure the mask has the right size.
7186 if (NumElts != M.size())
7187 return false;
7188
7189 // Look for <15, ..., 3, -1, 1, 0>.
7190 for (unsigned i = 0; i != NumElts; ++i)
7191 if (M[i] >= 0 && M[i] != (int) (NumElts - 1 - i))
7192 return false;
7193
7194 return true;
7195}
7196
7197static bool isTruncMask(ArrayRef<int> M, EVT VT, bool Top, bool SingleSource) {
7198 unsigned NumElts = VT.getVectorNumElements();
7199 // Make sure the mask has the right size.
7200 if (NumElts != M.size() || (VT != MVT::v8i16 && VT != MVT::v16i8))
7201 return false;
7202
7203 // Half-width truncation patterns (e.g. v4i32 -> v8i16):
7204 // !Top && SingleSource: <0, 2, 4, 6, 0, 2, 4, 6>
7205 // !Top && !SingleSource: <0, 2, 4, 6, 8, 10, 12, 14>
7206 // Top && SingleSource: <1, 3, 5, 7, 1, 3, 5, 7>
7207 // Top && !SingleSource: <1, 3, 5, 7, 9, 11, 13, 15>
7208 int Ofs = Top ? 1 : 0;
7209 int Upper = SingleSource ? 0 : NumElts;
7210 for (int i = 0, e = NumElts / 2; i != e; ++i) {
7211 if (M[i] >= 0 && M[i] != (i * 2) + Ofs)
7212 return false;
7213 if (M[i + e] >= 0 && M[i + e] != (i * 2) + Ofs + Upper)
7214 return false;
7215 }
7216 return true;
7217}
7218
7219static bool isVMOVNMask(ArrayRef<int> M, EVT VT, bool Top, bool SingleSource) {
7220 unsigned NumElts = VT.getVectorNumElements();
7221 // Make sure the mask has the right size.
7222 if (NumElts != M.size() || (VT != MVT::v8i16 && VT != MVT::v16i8))
7223 return false;
7224
7225 // If Top
7226 // Look for <0, N, 2, N+2, 4, N+4, ..>.
7227 // This inserts Input2 into Input1
7228 // else if not Top
7229 // Look for <0, N+1, 2, N+3, 4, N+5, ..>
7230 // This inserts Input1 into Input2
7231 unsigned Offset = Top ? 0 : 1;
7232 unsigned N = SingleSource ? 0 : NumElts;
7233 for (unsigned i = 0; i < NumElts; i += 2) {
7234 if (M[i] >= 0 && M[i] != (int)i)
7235 return false;
7236 if (M[i + 1] >= 0 && M[i + 1] != (int)(N + i + Offset))
7237 return false;
7238 }
7239
7240 return true;
7241}
7242
7243static bool isVMOVNTruncMask(ArrayRef<int> M, EVT ToVT, bool rev) {
7244 unsigned NumElts = ToVT.getVectorNumElements();
7245 if (NumElts != M.size())
7246 return false;
7247
7248 // Test if the Trunc can be convertible to a VMOVN with this shuffle. We are
7249 // looking for patterns of:
7250 // !rev: 0 N/2 1 N/2+1 2 N/2+2 ...
7251 // rev: N/2 0 N/2+1 1 N/2+2 2 ...
7252
7253 unsigned Off0 = rev ? NumElts / 2 : 0;
7254 unsigned Off1 = rev ? 0 : NumElts / 2;
7255 for (unsigned i = 0; i < NumElts; i += 2) {
7256 if (M[i] >= 0 && M[i] != (int)(Off0 + i / 2))
7257 return false;
7258 if (M[i + 1] >= 0 && M[i + 1] != (int)(Off1 + i / 2))
7259 return false;
7260 }
7261
7262 return true;
7263}
7264
7265// Reconstruct an MVE VCVT from a BuildVector of scalar fptrunc, all extracted
7266// from a pair of inputs. For example:
7267// BUILDVECTOR(FP_ROUND(EXTRACT_ELT(X, 0),
7268// FP_ROUND(EXTRACT_ELT(Y, 0),
7269// FP_ROUND(EXTRACT_ELT(X, 1),
7270// FP_ROUND(EXTRACT_ELT(Y, 1), ...)
7272 const ARMSubtarget *ST) {
7273 assert(BV.getOpcode() == ISD::BUILD_VECTOR && "Unknown opcode!");
7274 if (!ST->hasMVEFloatOps())
7275 return SDValue();
7276
7277 SDLoc dl(BV);
7278 EVT VT = BV.getValueType();
7279 if (VT != MVT::v8f16)
7280 return SDValue();
7281
7282 // We are looking for a buildvector of fptrunc elements, where all the
7283 // elements are interleavingly extracted from two sources. Check the first two
7284 // items are valid enough and extract some info from them (they are checked
7285 // properly in the loop below).
7286 if (BV.getOperand(0).getOpcode() != ISD::FP_ROUND ||
7289 return SDValue();
7290 if (BV.getOperand(1).getOpcode() != ISD::FP_ROUND ||
7293 return SDValue();
7294 SDValue Op0 = BV.getOperand(0).getOperand(0).getOperand(0);
7295 SDValue Op1 = BV.getOperand(1).getOperand(0).getOperand(0);
7296 if (Op0.getValueType() != MVT::v4f32 || Op1.getValueType() != MVT::v4f32)
7297 return SDValue();
7298
7299 // Check all the values in the BuildVector line up with our expectations.
7300 for (unsigned i = 1; i < 4; i++) {
7301 auto Check = [](SDValue Trunc, SDValue Op, unsigned Idx) {
7302 return Trunc.getOpcode() == ISD::FP_ROUND &&
7304 Trunc.getOperand(0).getOperand(0) == Op &&
7305 Trunc.getOperand(0).getConstantOperandVal(1) == Idx;
7306 };
7307 if (!Check(BV.getOperand(i * 2 + 0), Op0, i))
7308 return SDValue();
7309 if (!Check(BV.getOperand(i * 2 + 1), Op1, i))
7310 return SDValue();
7311 }
7312
7313 SDValue N1 = DAG.getNode(ARMISD::VCVTN, dl, VT, DAG.getUNDEF(VT), Op0,
7314 DAG.getConstant(0, dl, MVT::i32));
7315 return DAG.getNode(ARMISD::VCVTN, dl, VT, N1, Op1,
7316 DAG.getConstant(1, dl, MVT::i32));
7317}
7318
7319// Reconstruct an MVE VCVT from a BuildVector of scalar fpext, all extracted
7320// from a single input on alternating lanes. For example:
7321// BUILDVECTOR(FP_ROUND(EXTRACT_ELT(X, 0),
7322// FP_ROUND(EXTRACT_ELT(X, 2),
7323// FP_ROUND(EXTRACT_ELT(X, 4), ...)
7325 const ARMSubtarget *ST) {
7326 assert(BV.getOpcode() == ISD::BUILD_VECTOR && "Unknown opcode!");
7327 if (!ST->hasMVEFloatOps())
7328 return SDValue();
7329
7330 SDLoc dl(BV);
7331 EVT VT = BV.getValueType();
7332 if (VT != MVT::v4f32)
7333 return SDValue();
7334
7335 // We are looking for a buildvector of fptext elements, where all the
7336 // elements are alternating lanes from a single source. For example <0,2,4,6>
7337 // or <1,3,5,7>. Check the first two items are valid enough and extract some
7338 // info from them (they are checked properly in the loop below).
7339 if (BV.getOperand(0).getOpcode() != ISD::FP_EXTEND ||
7341 return SDValue();
7342 SDValue Op0 = BV.getOperand(0).getOperand(0).getOperand(0);
7344 if (Op0.getValueType() != MVT::v8f16 || (Offset != 0 && Offset != 1))
7345 return SDValue();
7346
7347 // Check all the values in the BuildVector line up with our expectations.
7348 for (unsigned i = 1; i < 4; i++) {
7349 auto Check = [](SDValue Trunc, SDValue Op, unsigned Idx) {
7350 return Trunc.getOpcode() == ISD::FP_EXTEND &&
7352 Trunc.getOperand(0).getOperand(0) == Op &&
7353 Trunc.getOperand(0).getConstantOperandVal(1) == Idx;
7354 };
7355 if (!Check(BV.getOperand(i), Op0, 2 * i + Offset))
7356 return SDValue();
7357 }
7358
7359 return DAG.getNode(ARMISD::VCVTL, dl, VT, Op0,
7360 DAG.getConstant(Offset, dl, MVT::i32));
7361}
7362
7363// If N is an integer constant that can be moved into a register in one
7364// instruction, return an SDValue of such a constant (will become a MOV
7365// instruction). Otherwise return null.
7367 const ARMSubtarget *ST, const SDLoc &dl) {
7368 uint64_t Val;
7369 if (!isa<ConstantSDNode>(N))
7370 return SDValue();
7371 Val = N->getAsZExtVal();
7372
7373 if (ST->isThumb1Only()) {
7374 if (Val <= 255 || ~Val <= 255)
7375 return DAG.getConstant(Val, dl, MVT::i32);
7376 } else {
7377 if (ARM_AM::getSOImmVal(Val) != -1 || ARM_AM::getSOImmVal(~Val) != -1)
7378 return DAG.getConstant(Val, dl, MVT::i32);
7379 }
7380 return SDValue();
7381}
7382
7384 const ARMSubtarget *ST) {
7385 SDLoc dl(Op);
7386 EVT VT = Op.getValueType();
7387
7388 assert(ST->hasMVEIntegerOps() && "LowerBUILD_VECTOR_i1 called without MVE!");
7389
7390 unsigned NumElts = VT.getVectorNumElements();
7391 unsigned BoolMask;
7392 unsigned BitsPerBool;
7393 if (NumElts == 2) {
7394 BitsPerBool = 8;
7395 BoolMask = 0xff;
7396 } else if (NumElts == 4) {
7397 BitsPerBool = 4;
7398 BoolMask = 0xf;
7399 } else if (NumElts == 8) {
7400 BitsPerBool = 2;
7401 BoolMask = 0x3;
7402 } else if (NumElts == 16) {
7403 BitsPerBool = 1;
7404 BoolMask = 0x1;
7405 } else
7406 return SDValue();
7407
7408 // If this is a single value copied into all lanes (a splat), we can just sign
7409 // extend that single value
7410 SDValue FirstOp = Op.getOperand(0);
7411 if (!isa<ConstantSDNode>(FirstOp) &&
7412 llvm::all_of(llvm::drop_begin(Op->ops()), [&FirstOp](const SDUse &U) {
7413 return U.get().isUndef() || U.get() == FirstOp;
7414 })) {
7415 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND_INREG, dl, MVT::i32, FirstOp,
7416 DAG.getValueType(MVT::i1));
7417 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, Op.getValueType(), Ext);
7418 }
7419
7420 // First create base with bits set where known
7421 unsigned Bits32 = 0;
7422 for (unsigned i = 0; i < NumElts; ++i) {
7423 SDValue V = Op.getOperand(i);
7424 if (!isa<ConstantSDNode>(V) && !V.isUndef())
7425 continue;
7426 bool BitSet = V.isUndef() ? false : V->getAsZExtVal();
7427 if (BitSet)
7428 Bits32 |= BoolMask << (i * BitsPerBool);
7429 }
7430
7431 // Add in unknown nodes
7432 SDValue Base = DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT,
7433 DAG.getConstant(Bits32, dl, MVT::i32));
7434 for (unsigned i = 0; i < NumElts; ++i) {
7435 SDValue V = Op.getOperand(i);
7436 if (isa<ConstantSDNode>(V) || V.isUndef())
7437 continue;
7438 Base = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Base, V,
7439 DAG.getConstant(i, dl, MVT::i32));
7440 }
7441
7442 return Base;
7443}
7444
7446 const ARMSubtarget *ST) {
7447 if (!ST->hasMVEIntegerOps())
7448 return SDValue();
7449
7450 // We are looking for a buildvector where each element is Op[0] + i*N
7451 EVT VT = Op.getValueType();
7452 SDValue Op0 = Op.getOperand(0);
7453 unsigned NumElts = VT.getVectorNumElements();
7454
7455 // Get the increment value from operand 1
7456 SDValue Op1 = Op.getOperand(1);
7457 if (Op1.getOpcode() != ISD::ADD || Op1.getOperand(0) != Op0 ||
7459 return SDValue();
7460 unsigned N = Op1.getConstantOperandVal(1);
7461 if (N != 1 && N != 2 && N != 4 && N != 8)
7462 return SDValue();
7463
7464 // Check that each other operand matches
7465 for (unsigned I = 2; I < NumElts; I++) {
7466 SDValue OpI = Op.getOperand(I);
7467 if (OpI.getOpcode() != ISD::ADD || OpI.getOperand(0) != Op0 ||
7469 OpI.getConstantOperandVal(1) != I * N)
7470 return SDValue();
7471 }
7472
7473 SDLoc DL(Op);
7474 return DAG.getNode(ARMISD::VIDUP, DL, DAG.getVTList(VT, MVT::i32), Op0,
7475 DAG.getConstant(N, DL, MVT::i32));
7476}
7477
7478// Returns true if the operation N can be treated as qr instruction variant at
7479// operand Op.
7480static bool IsQRMVEInstruction(const SDNode *N, const SDNode *Op) {
7481 switch (N->getOpcode()) {
7482 case ISD::ADD:
7483 case ISD::MUL:
7484 case ISD::SADDSAT:
7485 case ISD::UADDSAT:
7486 case ISD::AVGFLOORS:
7487 case ISD::AVGFLOORU:
7488 return true;
7489 case ISD::SUB:
7490 case ISD::SSUBSAT:
7491 case ISD::USUBSAT:
7492 return N->getOperand(1).getNode() == Op;
7494 switch (N->getConstantOperandVal(0)) {
7495 case Intrinsic::arm_mve_add_predicated:
7496 case Intrinsic::arm_mve_mul_predicated:
7497 case Intrinsic::arm_mve_qadd_predicated:
7498 case Intrinsic::arm_mve_vhadd:
7499 case Intrinsic::arm_mve_hadd_predicated:
7500 case Intrinsic::arm_mve_vqdmulh:
7501 case Intrinsic::arm_mve_qdmulh_predicated:
7502 case Intrinsic::arm_mve_vqrdmulh:
7503 case Intrinsic::arm_mve_qrdmulh_predicated:
7504 case Intrinsic::arm_mve_vqdmull:
7505 case Intrinsic::arm_mve_vqdmull_predicated:
7506 return true;
7507 case Intrinsic::arm_mve_sub_predicated:
7508 case Intrinsic::arm_mve_qsub_predicated:
7509 case Intrinsic::arm_mve_vhsub:
7510 case Intrinsic::arm_mve_hsub_predicated:
7511 return N->getOperand(2).getNode() == Op;
7512 default:
7513 return false;
7514 }
7515 default:
7516 return false;
7517 }
7518}
7519
7520// If this is a case we can't handle, return null and let the default
7521// expansion code take care of it.
7522SDValue ARMTargetLowering::LowerBUILD_VECTOR(SDValue Op, SelectionDAG &DAG,
7523 const ARMSubtarget *ST) const {
7524 BuildVectorSDNode *BVN = cast<BuildVectorSDNode>(Op.getNode());
7525 SDLoc dl(Op);
7526 EVT VT = Op.getValueType();
7527
7528 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1)
7529 return LowerBUILD_VECTOR_i1(Op, DAG, ST);
7530
7531 if (SDValue R = LowerBUILD_VECTORToVIDUP(Op, DAG, ST))
7532 return R;
7533
7534 APInt SplatBits, SplatUndef;
7535 unsigned SplatBitSize;
7536 bool HasAnyUndefs;
7537 if (BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) {
7538 if (SplatUndef.isAllOnes())
7539 return DAG.getUNDEF(VT);
7540
7541 // If all the users of this constant splat are qr instruction variants,
7542 // generate a vdup of the constant.
7543 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == SplatBitSize &&
7544 (SplatBitSize == 8 || SplatBitSize == 16 || SplatBitSize == 32) &&
7545 all_of(BVN->users(),
7546 [BVN](const SDNode *U) { return IsQRMVEInstruction(U, BVN); })) {
7547 EVT DupVT = SplatBitSize == 32 ? MVT::v4i32
7548 : SplatBitSize == 16 ? MVT::v8i16
7549 : MVT::v16i8;
7550 SDValue Const = DAG.getConstant(SplatBits.getZExtValue(), dl, MVT::i32);
7551 SDValue VDup = DAG.getNode(ARMISD::VDUP, dl, DupVT, Const);
7552 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, VDup);
7553 }
7554
7555 if ((ST->hasNEON() && SplatBitSize <= 64) ||
7556 (ST->hasMVEIntegerOps() && SplatBitSize <= 64)) {
7557 // Check if an immediate VMOV works.
7558 EVT VmovVT;
7559 SDValue Val =
7560 isVMOVModifiedImm(SplatBits.getZExtValue(), SplatUndef.getZExtValue(),
7561 SplatBitSize, DAG, dl, VmovVT, VT, VMOVModImm);
7562
7563 if (Val.getNode()) {
7564 SDValue Vmov = DAG.getNode(ARMISD::VMOVIMM, dl, VmovVT, Val);
7565 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Vmov);
7566 }
7567
7568 // Try an immediate VMVN.
7569 uint64_t NegatedImm = (~SplatBits).getZExtValue();
7570 Val = isVMOVModifiedImm(
7571 NegatedImm, SplatUndef.getZExtValue(), SplatBitSize, DAG, dl, VmovVT,
7572 VT, ST->hasMVEIntegerOps() ? MVEVMVNModImm : VMVNModImm);
7573 if (Val.getNode()) {
7574 SDValue Vmov = DAG.getNode(ARMISD::VMVNIMM, dl, VmovVT, Val);
7575 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Vmov);
7576 }
7577
7578 // Use vmov.f32 to materialize other v2f32 and v4f32 splats.
7579 if ((VT == MVT::v2f32 || VT == MVT::v4f32) && SplatBitSize == 32) {
7580 int ImmVal = ARM_AM::getFP32Imm(SplatBits);
7581 if (ImmVal != -1) {
7582 SDValue Val = DAG.getTargetConstant(ImmVal, dl, MVT::i32);
7583 return DAG.getNode(ARMISD::VMOVFPIMM, dl, VT, Val);
7584 }
7585 }
7586
7587 // If we are under MVE, generate a VDUP(constant), bitcast to the original
7588 // type.
7589 if (ST->hasMVEIntegerOps() &&
7590 (SplatBitSize == 8 || SplatBitSize == 16 || SplatBitSize == 32)) {
7591 EVT DupVT = SplatBitSize == 32 ? MVT::v4i32
7592 : SplatBitSize == 16 ? MVT::v8i16
7593 : MVT::v16i8;
7594 SDValue Const = DAG.getConstant(SplatBits.getZExtValue(), dl, MVT::i32);
7595 SDValue VDup = DAG.getNode(ARMISD::VDUP, dl, DupVT, Const);
7596 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, VDup);
7597 }
7598 }
7599 }
7600
7601 // Scan through the operands to see if only one value is used.
7602 //
7603 // As an optimisation, even if more than one value is used it may be more
7604 // profitable to splat with one value then change some lanes.
7605 //
7606 // Heuristically we decide to do this if the vector has a "dominant" value,
7607 // defined as splatted to more than half of the lanes.
7608 unsigned NumElts = VT.getVectorNumElements();
7609 bool isOnlyLowElement = true;
7610 bool usesOnlyOneValue = true;
7611 bool hasDominantValue = false;
7612 bool isConstant = true;
7613
7614 // Map of the number of times a particular SDValue appears in the
7615 // element list.
7616 DenseMap<SDValue, unsigned> ValueCounts;
7617 SDValue Value;
7618 for (unsigned i = 0; i < NumElts; ++i) {
7619 SDValue V = Op.getOperand(i);
7620 if (V.isUndef())
7621 continue;
7622 if (i > 0)
7623 isOnlyLowElement = false;
7625 isConstant = false;
7626
7627 unsigned &Count = ValueCounts[V];
7628
7629 // Is this value dominant? (takes up more than half of the lanes)
7630 if (++Count > (NumElts / 2)) {
7631 hasDominantValue = true;
7632 Value = V;
7633 }
7634 }
7635 if (ValueCounts.size() != 1)
7636 usesOnlyOneValue = false;
7637 if (!Value.getNode() && !ValueCounts.empty())
7638 Value = ValueCounts.begin()->first;
7639
7640 if (ValueCounts.empty())
7641 return DAG.getUNDEF(VT);
7642
7643 // Loads are better lowered with insert_vector_elt/ARMISD::BUILD_VECTOR.
7644 // Keep going if we are hitting this case.
7645 if (isOnlyLowElement && !ISD::isNormalLoad(Value.getNode()) &&
7646 (VT != MVT::v8f16 || ST->hasFullFP16()))
7647 return DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, VT, Value);
7648
7649 unsigned EltSize = VT.getScalarSizeInBits();
7650
7651 // Use VDUP for non-constant splats. For f32 constant splats, reduce to
7652 // i32 and try again.
7653 if (hasDominantValue && EltSize <= 32) {
7654 if (!isConstant) {
7655 SDValue N;
7656
7657 // If we are VDUPing a value that comes directly from a vector, that will
7658 // cause an unnecessary move to and from a GPR, where instead we could
7659 // just use VDUPLANE. We can only do this if the lane being extracted
7660 // is at a constant index, as the VDUP from lane instructions only have
7661 // constant-index forms.
7662 ConstantSDNode *constIndex;
7663 if (Value->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7664 (constIndex = dyn_cast<ConstantSDNode>(Value->getOperand(1)))) {
7665 // We need to create a new undef vector to use for the VDUPLANE if the
7666 // size of the vector from which we get the value is different than the
7667 // size of the vector that we need to create. We will insert the element
7668 // such that the register coalescer will remove unnecessary copies.
7669 if (VT != Value->getOperand(0).getValueType()) {
7670 unsigned index = constIndex->getAPIntValue().getLimitedValue() %
7672 N = DAG.getNode(ARMISD::VDUPLANE, dl, VT,
7673 DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, DAG.getUNDEF(VT),
7674 Value, DAG.getConstant(index, dl, MVT::i32)),
7675 DAG.getConstant(index, dl, MVT::i32));
7676 } else
7677 N = DAG.getNode(ARMISD::VDUPLANE, dl, VT,
7678 Value->getOperand(0), Value->getOperand(1));
7679 } else
7680 N = DAG.getNode(ARMISD::VDUP, dl, VT, Value);
7681
7682 if (!usesOnlyOneValue) {
7683 // The dominant value was splatted as 'N', but we now have to insert
7684 // all differing elements.
7685 for (unsigned I = 0; I < NumElts; ++I) {
7686 if (Op.getOperand(I) == Value)
7687 continue;
7689 Ops.push_back(N);
7690 Ops.push_back(Op.getOperand(I));
7691 Ops.push_back(DAG.getConstant(I, dl, MVT::i32));
7692 N = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Ops);
7693 }
7694 }
7695 return N;
7696 }
7699 MVT FVT = VT.getVectorElementType().getSimpleVT();
7700 assert(FVT == MVT::f32 || FVT == MVT::f16);
7701 MVT IVT = (FVT == MVT::f32) ? MVT::i32 : MVT::i16;
7702 for (unsigned i = 0; i < NumElts; ++i)
7703 Ops.push_back(DAG.getNode(ISD::BITCAST, dl, IVT,
7704 Op.getOperand(i)));
7705 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), IVT, NumElts);
7706 SDValue Val = DAG.getBuildVector(VecVT, dl, Ops);
7707 Val = LowerBUILD_VECTOR(Val, DAG, ST);
7708 if (Val.getNode())
7709 return DAG.getNode(ISD::BITCAST, dl, VT, Val);
7710 }
7711 if (usesOnlyOneValue) {
7712 SDValue Val = IsSingleInstrConstant(Value, DAG, ST, dl);
7713 if (isConstant && Val.getNode())
7714 return DAG.getNode(ARMISD::VDUP, dl, VT, Val);
7715 }
7716 }
7717
7718 // If all elements are constants and the case above didn't get hit, fall back
7719 // to the default expansion, which will generate a load from the constant
7720 // pool.
7721 if (isConstant)
7722 return SDValue();
7723
7724 // Reconstruct the BUILDVECTOR to one of the legal shuffles (such as vext and
7725 // vmovn). Empirical tests suggest this is rarely worth it for vectors of
7726 // length <= 2.
7727 if (NumElts >= 4)
7728 if (SDValue shuffle = ReconstructShuffle(Op, DAG))
7729 return shuffle;
7730
7731 // Attempt to turn a buildvector of scalar fptrunc's or fpext's back into
7732 // VCVT's
7733 if (SDValue VCVT = LowerBuildVectorOfFPTrunc(Op, DAG, Subtarget))
7734 return VCVT;
7735 if (SDValue VCVT = LowerBuildVectorOfFPExt(Op, DAG, Subtarget))
7736 return VCVT;
7737
7738 if (ST->hasNEON() && VT.is128BitVector() && VT != MVT::v2f64 && VT != MVT::v4f32) {
7739 // If we haven't found an efficient lowering, try splitting a 128-bit vector
7740 // into two 64-bit vectors; we might discover a better way to lower it.
7741 SmallVector<SDValue, 64> Ops(Op->op_begin(), Op->op_begin() + NumElts);
7742 EVT ExtVT = VT.getVectorElementType();
7743 EVT HVT = EVT::getVectorVT(*DAG.getContext(), ExtVT, NumElts / 2);
7744 SDValue Lower = DAG.getBuildVector(HVT, dl, ArrayRef(&Ops[0], NumElts / 2));
7745 if (Lower.getOpcode() == ISD::BUILD_VECTOR)
7746 Lower = LowerBUILD_VECTOR(Lower, DAG, ST);
7747 SDValue Upper =
7748 DAG.getBuildVector(HVT, dl, ArrayRef(&Ops[NumElts / 2], NumElts / 2));
7749 if (Upper.getOpcode() == ISD::BUILD_VECTOR)
7750 Upper = LowerBUILD_VECTOR(Upper, DAG, ST);
7751 if (Lower && Upper)
7752 return DAG.getNode(ISD::CONCAT_VECTORS, dl, VT, Lower, Upper);
7753 }
7754
7755 // Vectors with 32- or 64-bit elements can be built by directly assigning
7756 // the subregisters. Lower it to an ARMISD::BUILD_VECTOR so the operands
7757 // will be legalized.
7758 if (EltSize >= 32) {
7759 // Do the expansion with floating-point types, since that is what the VFP
7760 // registers are defined to use, and since i64 is not legal.
7761 EVT EltVT = EVT::getFloatingPointVT(EltSize);
7762 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts);
7764 for (unsigned i = 0; i < NumElts; ++i)
7765 Ops.push_back(DAG.getNode(ISD::BITCAST, dl, EltVT, Op.getOperand(i)));
7766 SDValue Val = DAG.getNode(ARMISD::BUILD_VECTOR, dl, VecVT, Ops);
7767 return DAG.getNode(ISD::BITCAST, dl, VT, Val);
7768 }
7769
7770 // If all else fails, just use a sequence of INSERT_VECTOR_ELT when we
7771 // know the default expansion would otherwise fall back on something even
7772 // worse. For a vector with one or two non-undef values, that's
7773 // scalar_to_vector for the elements followed by a shuffle (provided the
7774 // shuffle is valid for the target) and materialization element by element
7775 // on the stack followed by a load for everything else.
7776 if ((!isConstant && !usesOnlyOneValue) ||
7777 (VT == MVT::v8f16 && !ST->hasFullFP16())) {
7778 SDValue Vec = DAG.getUNDEF(VT);
7779 for (unsigned i = 0 ; i < NumElts; ++i) {
7780 SDValue V = Op.getOperand(i);
7781 if (V.isUndef())
7782 continue;
7783 SDValue LaneIdx = DAG.getConstant(i, dl, MVT::i32);
7784 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, Vec, V, LaneIdx);
7785 }
7786 return Vec;
7787 }
7788
7789 return SDValue();
7790}
7791
7792// Gather data to see if the operation can be modelled as a
7793// shuffle in combination with VEXTs.
7794SDValue ARMTargetLowering::ReconstructShuffle(SDValue Op,
7795 SelectionDAG &DAG) const {
7796 assert(Op.getOpcode() == ISD::BUILD_VECTOR && "Unknown opcode!");
7797 SDLoc dl(Op);
7798 EVT VT = Op.getValueType();
7799 unsigned NumElts = VT.getVectorNumElements();
7800
7801 struct ShuffleSourceInfo {
7802 SDValue Vec;
7803 unsigned MinElt = std::numeric_limits<unsigned>::max();
7804 unsigned MaxElt = 0;
7805
7806 // We may insert some combination of BITCASTs and VEXT nodes to force Vec to
7807 // be compatible with the shuffle we intend to construct. As a result
7808 // ShuffleVec will be some sliding window into the original Vec.
7809 SDValue ShuffleVec;
7810
7811 // Code should guarantee that element i in Vec starts at element "WindowBase
7812 // + i * WindowScale in ShuffleVec".
7813 int WindowBase = 0;
7814 int WindowScale = 1;
7815
7816 ShuffleSourceInfo(SDValue Vec) : Vec(Vec), ShuffleVec(Vec) {}
7817
7818 bool operator ==(SDValue OtherVec) { return Vec == OtherVec; }
7819 };
7820
7821 // First gather all vectors used as an immediate source for this BUILD_VECTOR
7822 // node.
7824 for (unsigned i = 0; i < NumElts; ++i) {
7825 SDValue V = Op.getOperand(i);
7826 if (V.isUndef())
7827 continue;
7828 else if (V.getOpcode() != ISD::EXTRACT_VECTOR_ELT) {
7829 // A shuffle can only come from building a vector from various
7830 // elements of other vectors.
7831 return SDValue();
7832 } else if (!isa<ConstantSDNode>(V.getOperand(1))) {
7833 // Furthermore, shuffles require a constant mask, whereas extractelts
7834 // accept variable indices.
7835 return SDValue();
7836 }
7837
7838 // Add this element source to the list if it's not already there.
7839 SDValue SourceVec = V.getOperand(0);
7840 auto Source = llvm::find(Sources, SourceVec);
7841 if (Source == Sources.end())
7842 Source = Sources.insert(Sources.end(), ShuffleSourceInfo(SourceVec));
7843
7844 // Update the minimum and maximum lane number seen.
7845 unsigned EltNo = V.getConstantOperandVal(1);
7846 Source->MinElt = std::min(Source->MinElt, EltNo);
7847 Source->MaxElt = std::max(Source->MaxElt, EltNo);
7848 }
7849
7850 // Currently only do something sane when at most two source vectors
7851 // are involved.
7852 if (Sources.size() > 2)
7853 return SDValue();
7854
7855 // Find out the smallest element size among result and two sources, and use
7856 // it as element size to build the shuffle_vector.
7857 EVT SmallestEltTy = VT.getVectorElementType();
7858 for (auto &Source : Sources) {
7859 EVT SrcEltTy = Source.Vec.getValueType().getVectorElementType();
7860 if (SrcEltTy.bitsLT(SmallestEltTy))
7861 SmallestEltTy = SrcEltTy;
7862 }
7863 unsigned ResMultiplier =
7864 VT.getScalarSizeInBits() / SmallestEltTy.getSizeInBits();
7865 NumElts = VT.getSizeInBits() / SmallestEltTy.getSizeInBits();
7866 EVT ShuffleVT = EVT::getVectorVT(*DAG.getContext(), SmallestEltTy, NumElts);
7867
7868 // If the source vector is too wide or too narrow, we may nevertheless be able
7869 // to construct a compatible shuffle either by concatenating it with UNDEF or
7870 // extracting a suitable range of elements.
7871 for (auto &Src : Sources) {
7872 EVT SrcVT = Src.ShuffleVec.getValueType();
7873
7874 uint64_t SrcVTSize = SrcVT.getFixedSizeInBits();
7875 uint64_t VTSize = VT.getFixedSizeInBits();
7876 if (SrcVTSize == VTSize)
7877 continue;
7878
7879 // This stage of the search produces a source with the same element type as
7880 // the original, but with a total width matching the BUILD_VECTOR output.
7881 EVT EltVT = SrcVT.getVectorElementType();
7882 unsigned NumSrcElts = VTSize / EltVT.getFixedSizeInBits();
7883 EVT DestVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumSrcElts);
7884
7885 if (SrcVTSize < VTSize) {
7886 if (2 * SrcVTSize != VTSize)
7887 return SDValue();
7888 // We can pad out the smaller vector for free, so if it's part of a
7889 // shuffle...
7890 Src.ShuffleVec =
7891 DAG.getNode(ISD::CONCAT_VECTORS, dl, DestVT, Src.ShuffleVec,
7892 DAG.getUNDEF(Src.ShuffleVec.getValueType()));
7893 continue;
7894 }
7895
7896 if (SrcVTSize != 2 * VTSize)
7897 return SDValue();
7898
7899 if (Src.MaxElt - Src.MinElt >= NumSrcElts) {
7900 // Span too large for a VEXT to cope
7901 return SDValue();
7902 }
7903
7904 if (Src.MinElt >= NumSrcElts) {
7905 // The extraction can just take the second half
7906 Src.ShuffleVec =
7907 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec,
7908 DAG.getConstant(NumSrcElts, dl, MVT::i32));
7909 Src.WindowBase = -NumSrcElts;
7910 } else if (Src.MaxElt < NumSrcElts) {
7911 // The extraction can just take the first half
7912 Src.ShuffleVec =
7913 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec,
7914 DAG.getConstant(0, dl, MVT::i32));
7915 } else {
7916 // An actual VEXT is needed
7917 SDValue VEXTSrc1 =
7918 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec,
7919 DAG.getConstant(0, dl, MVT::i32));
7920 SDValue VEXTSrc2 =
7921 DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, DestVT, Src.ShuffleVec,
7922 DAG.getConstant(NumSrcElts, dl, MVT::i32));
7923
7924 Src.ShuffleVec = DAG.getNode(ARMISD::VEXT, dl, DestVT, VEXTSrc1,
7925 VEXTSrc2,
7926 DAG.getConstant(Src.MinElt, dl, MVT::i32));
7927 Src.WindowBase = -Src.MinElt;
7928 }
7929 }
7930
7931 // Another possible incompatibility occurs from the vector element types. We
7932 // can fix this by bitcasting the source vectors to the same type we intend
7933 // for the shuffle.
7934 for (auto &Src : Sources) {
7935 EVT SrcEltTy = Src.ShuffleVec.getValueType().getVectorElementType();
7936 if (SrcEltTy == SmallestEltTy)
7937 continue;
7938 assert(ShuffleVT.getVectorElementType() == SmallestEltTy);
7939 Src.ShuffleVec = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, ShuffleVT, Src.ShuffleVec);
7940 Src.WindowScale = SrcEltTy.getSizeInBits() / SmallestEltTy.getSizeInBits();
7941 Src.WindowBase *= Src.WindowScale;
7942 }
7943
7944 // Final check before we try to actually produce a shuffle.
7945 LLVM_DEBUG({
7946 for (auto Src : Sources)
7947 assert(Src.ShuffleVec.getValueType() == ShuffleVT);
7948 });
7949
7950 // The stars all align, our next step is to produce the mask for the shuffle.
7951 SmallVector<int, 8> Mask(ShuffleVT.getVectorNumElements(), -1);
7952 int BitsPerShuffleLane = ShuffleVT.getScalarSizeInBits();
7953 for (unsigned i = 0; i < VT.getVectorNumElements(); ++i) {
7954 SDValue Entry = Op.getOperand(i);
7955 if (Entry.isUndef())
7956 continue;
7957
7958 auto Src = llvm::find(Sources, Entry.getOperand(0));
7959 int EltNo = cast<ConstantSDNode>(Entry.getOperand(1))->getSExtValue();
7960
7961 // EXTRACT_VECTOR_ELT performs an implicit any_ext; BUILD_VECTOR an implicit
7962 // trunc. So only std::min(SrcBits, DestBits) actually get defined in this
7963 // segment.
7964 EVT OrigEltTy = Entry.getOperand(0).getValueType().getVectorElementType();
7965 int BitsDefined = std::min(OrigEltTy.getScalarSizeInBits(),
7966 VT.getScalarSizeInBits());
7967 int LanesDefined = BitsDefined / BitsPerShuffleLane;
7968
7969 // This source is expected to fill ResMultiplier lanes of the final shuffle,
7970 // starting at the appropriate offset.
7971 int *LaneMask = &Mask[i * ResMultiplier];
7972
7973 int ExtractBase = EltNo * Src->WindowScale + Src->WindowBase;
7974 ExtractBase += NumElts * (Src - Sources.begin());
7975 for (int j = 0; j < LanesDefined; ++j)
7976 LaneMask[j] = ExtractBase + j;
7977 }
7978
7979
7980 // We can't handle more than two sources. This should have already
7981 // been checked before this point.
7982 assert(Sources.size() <= 2 && "Too many sources!");
7983
7984 SDValue ShuffleOps[] = { DAG.getUNDEF(ShuffleVT), DAG.getUNDEF(ShuffleVT) };
7985 for (unsigned i = 0; i < Sources.size(); ++i)
7986 ShuffleOps[i] = Sources[i].ShuffleVec;
7987
7988 SDValue Shuffle = buildLegalVectorShuffle(ShuffleVT, dl, ShuffleOps[0],
7989 ShuffleOps[1], Mask, DAG);
7990 if (!Shuffle)
7991 return SDValue();
7992 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Shuffle);
7993}
7994
7996 OP_COPY = 0, // Copy, used for things like <u,u,u,3> to say it is <0,1,2,3>
8005 OP_VUZPL, // VUZP, left result
8006 OP_VUZPR, // VUZP, right result
8007 OP_VZIPL, // VZIP, left result
8008 OP_VZIPR, // VZIP, right result
8009 OP_VTRNL, // VTRN, left result
8010 OP_VTRNR // VTRN, right result
8011};
8012
8013static bool isLegalMVEShuffleOp(unsigned PFEntry) {
8014 unsigned OpNum = (PFEntry >> 26) & 0x0F;
8015 switch (OpNum) {
8016 case OP_COPY:
8017 case OP_VREV:
8018 case OP_VDUP0:
8019 case OP_VDUP1:
8020 case OP_VDUP2:
8021 case OP_VDUP3:
8022 return true;
8023 }
8024 return false;
8025}
8026
8027/// isShuffleMaskLegal - Targets can use this to indicate that they only
8028/// support *some* VECTOR_SHUFFLE operations, those with specific masks.
8029/// By default, if a target supports the VECTOR_SHUFFLE node, all mask values
8030/// are assumed to be legal.
8032 if (VT.getVectorNumElements() == 4 &&
8033 (VT.is128BitVector() || VT.is64BitVector())) {
8034 unsigned PFIndexes[4];
8035 for (unsigned i = 0; i != 4; ++i) {
8036 if (M[i] < 0)
8037 PFIndexes[i] = 8;
8038 else
8039 PFIndexes[i] = M[i];
8040 }
8041
8042 // Compute the index in the perfect shuffle table.
8043 unsigned PFTableIndex =
8044 PFIndexes[0]*9*9*9+PFIndexes[1]*9*9+PFIndexes[2]*9+PFIndexes[3];
8045 unsigned PFEntry = PerfectShuffleTable[PFTableIndex];
8046 unsigned Cost = (PFEntry >> 30);
8047
8048 if (Cost <= 4 && (Subtarget->hasNEON() || isLegalMVEShuffleOp(PFEntry)))
8049 return true;
8050 }
8051
8052 bool ReverseVEXT, isV_UNDEF;
8053 unsigned Imm, WhichResult;
8054
8055 unsigned EltSize = VT.getScalarSizeInBits();
8056 if (EltSize >= 32 ||
8058 ShuffleVectorInst::isIdentityMask(M, M.size()) ||
8059 isVREVMask(M, VT, 64) ||
8060 isVREVMask(M, VT, 32) ||
8061 isVREVMask(M, VT, 16))
8062 return true;
8063 else if (Subtarget->hasNEON() &&
8064 (isVEXTMask(M, VT, ReverseVEXT, Imm) ||
8065 isVTBLMask(M, VT) ||
8066 isNEONTwoResultShuffleMask(M, VT, WhichResult, isV_UNDEF)))
8067 return true;
8068 else if ((VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v16i8) &&
8069 isReverseMask(M, VT))
8070 return true;
8071 else if (Subtarget->hasMVEIntegerOps() &&
8072 (isVMOVNMask(M, VT, true, false) ||
8073 isVMOVNMask(M, VT, false, false) || isVMOVNMask(M, VT, true, true)))
8074 return true;
8075 else if (Subtarget->hasMVEIntegerOps() &&
8076 (isTruncMask(M, VT, false, false) ||
8077 isTruncMask(M, VT, false, true) ||
8078 isTruncMask(M, VT, true, false) || isTruncMask(M, VT, true, true)))
8079 return true;
8080 else
8081 return false;
8082}
8083
8084/// GeneratePerfectShuffle - Given an entry in the perfect-shuffle table, emit
8085/// the specified operations to build the shuffle.
8087 SDValue RHS, SelectionDAG &DAG,
8088 const SDLoc &dl) {
8089 unsigned OpNum = (PFEntry >> 26) & 0x0F;
8090 unsigned LHSID = (PFEntry >> 13) & ((1 << 13)-1);
8091 unsigned RHSID = (PFEntry >> 0) & ((1 << 13)-1);
8092
8093 if (OpNum == OP_COPY) {
8094 if (LHSID == (1*9+2)*9+3) return LHS;
8095 assert(LHSID == ((4*9+5)*9+6)*9+7 && "Illegal OP_COPY!");
8096 return RHS;
8097 }
8098
8099 SDValue OpLHS, OpRHS;
8100 OpLHS = GeneratePerfectShuffle(PerfectShuffleTable[LHSID], LHS, RHS, DAG, dl);
8101 OpRHS = GeneratePerfectShuffle(PerfectShuffleTable[RHSID], LHS, RHS, DAG, dl);
8102 EVT VT = OpLHS.getValueType();
8103
8104 switch (OpNum) {
8105 default: llvm_unreachable("Unknown shuffle opcode!");
8106 case OP_VREV:
8107 // VREV divides the vector in half and swaps within the half.
8108 if (VT.getScalarSizeInBits() == 32)
8109 return DAG.getNode(ARMISD::VREV64, dl, VT, OpLHS);
8110 // vrev <4 x i16> -> VREV32
8111 if (VT.getScalarSizeInBits() == 16)
8112 return DAG.getNode(ARMISD::VREV32, dl, VT, OpLHS);
8113 // vrev <4 x i8> -> VREV16
8114 assert(VT.getScalarSizeInBits() == 8);
8115 return DAG.getNode(ARMISD::VREV16, dl, VT, OpLHS);
8116 case OP_VDUP0:
8117 case OP_VDUP1:
8118 case OP_VDUP2:
8119 case OP_VDUP3:
8120 return DAG.getNode(ARMISD::VDUPLANE, dl, VT,
8121 OpLHS, DAG.getConstant(OpNum-OP_VDUP0, dl, MVT::i32));
8122 case OP_VEXT1:
8123 case OP_VEXT2:
8124 case OP_VEXT3:
8125 return DAG.getNode(ARMISD::VEXT, dl, VT,
8126 OpLHS, OpRHS,
8127 DAG.getConstant(OpNum - OP_VEXT1 + 1, dl, MVT::i32));
8128 case OP_VUZPL:
8129 case OP_VUZPR:
8130 return DAG.getNode(ARMISD::VUZP, dl, DAG.getVTList(VT, VT),
8131 OpLHS, OpRHS).getValue(OpNum-OP_VUZPL);
8132 case OP_VZIPL:
8133 case OP_VZIPR:
8134 return DAG.getNode(ARMISD::VZIP, dl, DAG.getVTList(VT, VT),
8135 OpLHS, OpRHS).getValue(OpNum-OP_VZIPL);
8136 case OP_VTRNL:
8137 case OP_VTRNR:
8138 return DAG.getNode(ARMISD::VTRN, dl, DAG.getVTList(VT, VT),
8139 OpLHS, OpRHS).getValue(OpNum-OP_VTRNL);
8140 }
8141}
8142
8144 ArrayRef<int> ShuffleMask,
8145 SelectionDAG &DAG) {
8146 // Check to see if we can use the VTBL instruction.
8147 SDValue V1 = Op.getOperand(0);
8148 SDValue V2 = Op.getOperand(1);
8149 SDLoc DL(Op);
8150
8151 SmallVector<SDValue, 8> VTBLMask;
8152 for (int I : ShuffleMask)
8153 VTBLMask.push_back(DAG.getSignedConstant(I, DL, MVT::i32));
8154
8155 if (V2.getNode()->isUndef())
8156 return DAG.getNode(ARMISD::VTBL1, DL, MVT::v8i8, V1,
8157 DAG.getBuildVector(MVT::v8i8, DL, VTBLMask));
8158
8159 return DAG.getNode(ARMISD::VTBL2, DL, MVT::v8i8, V1, V2,
8160 DAG.getBuildVector(MVT::v8i8, DL, VTBLMask));
8161}
8162
8164 SDLoc DL(Op);
8165 EVT VT = Op.getValueType();
8166
8167 assert((VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v16i8) &&
8168 "Expect an v8i16/v16i8 type");
8169 SDValue OpLHS = DAG.getNode(ARMISD::VREV64, DL, VT, Op.getOperand(0));
8170 // For a v16i8 type: After the VREV, we have got <7, ..., 0, 15, ..., 8>. Now,
8171 // extract the first 8 bytes into the top double word and the last 8 bytes
8172 // into the bottom double word, through a new vector shuffle that will be
8173 // turned into a VEXT on Neon, or a couple of VMOVDs on MVE.
8174 std::vector<int> NewMask;
8175 for (unsigned i = 0; i < VT.getVectorNumElements() / 2; i++)
8176 NewMask.push_back(VT.getVectorNumElements() / 2 + i);
8177 for (unsigned i = 0; i < VT.getVectorNumElements() / 2; i++)
8178 NewMask.push_back(i);
8179 return DAG.getVectorShuffle(VT, DL, OpLHS, OpLHS, NewMask);
8180}
8181
8183 switch (VT.getSimpleVT().SimpleTy) {
8184 case MVT::v2i1:
8185 return MVT::v2f64;
8186 case MVT::v4i1:
8187 return MVT::v4i32;
8188 case MVT::v8i1:
8189 return MVT::v8i16;
8190 case MVT::v16i1:
8191 return MVT::v16i8;
8192 default:
8193 llvm_unreachable("Unexpected vector predicate type");
8194 }
8195}
8196
8198 SelectionDAG &DAG) {
8199 // Converting from boolean predicates to integers involves creating a vector
8200 // of all ones or all zeroes and selecting the lanes based upon the real
8201 // predicate.
8203 DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0xff), dl, MVT::i32);
8204 AllOnes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v16i8, AllOnes);
8205
8206 SDValue AllZeroes =
8207 DAG.getTargetConstant(ARM_AM::createVMOVModImm(0xe, 0x0), dl, MVT::i32);
8208 AllZeroes = DAG.getNode(ARMISD::VMOVIMM, dl, MVT::v16i8, AllZeroes);
8209
8210 // Get full vector type from predicate type
8212
8213 SDValue RecastV1;
8214 // If the real predicate is an v8i1 or v4i1 (not v16i1) then we need to recast
8215 // this to a v16i1. This cannot be done with an ordinary bitcast because the
8216 // sizes are not the same. We have to use a MVE specific PREDICATE_CAST node,
8217 // since we know in hardware the sizes are really the same.
8218 if (VT != MVT::v16i1)
8219 RecastV1 = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v16i1, Pred);
8220 else
8221 RecastV1 = Pred;
8222
8223 // Select either all ones or zeroes depending upon the real predicate bits.
8224 SDValue PredAsVector =
8225 DAG.getNode(ISD::VSELECT, dl, MVT::v16i8, RecastV1, AllOnes, AllZeroes);
8226
8227 // Recast our new predicate-as-integer v16i8 vector into something
8228 // appropriate for the shuffle, i.e. v4i32 for a real v4i1 predicate.
8229 return DAG.getNode(ISD::BITCAST, dl, NewVT, PredAsVector);
8230}
8231
8233 const ARMSubtarget *ST) {
8234 EVT VT = Op.getValueType();
8236 ArrayRef<int> ShuffleMask = SVN->getMask();
8237
8238 assert(ST->hasMVEIntegerOps() &&
8239 "No support for vector shuffle of boolean predicates");
8240
8241 SDValue V1 = Op.getOperand(0);
8242 SDValue V2 = Op.getOperand(1);
8243 SDLoc dl(Op);
8244 if (isReverseMask(ShuffleMask, VT)) {
8245 SDValue cast = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, V1);
8246 SDValue rbit = DAG.getNode(ISD::BITREVERSE, dl, MVT::i32, cast);
8247 SDValue srl = DAG.getNode(ISD::SRL, dl, MVT::i32, rbit,
8248 DAG.getConstant(16, dl, MVT::i32));
8249 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT, srl);
8250 }
8251
8252 // Until we can come up with optimised cases for every single vector
8253 // shuffle in existence we have chosen the least painful strategy. This is
8254 // to essentially promote the boolean predicate to a 8-bit integer, where
8255 // each predicate represents a byte. Then we fall back on a normal integer
8256 // vector shuffle and convert the result back into a predicate vector. In
8257 // many cases the generated code might be even better than scalar code
8258 // operating on bits. Just imagine trying to shuffle 8 arbitrary 2-bit
8259 // fields in a register into 8 other arbitrary 2-bit fields!
8260 SDValue PredAsVector1 = PromoteMVEPredVector(dl, V1, VT, DAG);
8261 EVT NewVT = PredAsVector1.getValueType();
8262 SDValue PredAsVector2 = V2.isUndef() ? DAG.getUNDEF(NewVT)
8263 : PromoteMVEPredVector(dl, V2, VT, DAG);
8264 assert(PredAsVector2.getValueType() == NewVT &&
8265 "Expected identical vector type in expanded i1 shuffle!");
8266
8267 // Do the shuffle!
8268 SDValue Shuffled = DAG.getVectorShuffle(NewVT, dl, PredAsVector1,
8269 PredAsVector2, ShuffleMask);
8270
8271 // Now return the result of comparing the shuffled vector with zero,
8272 // which will generate a real predicate, i.e. v4i1, v8i1 or v16i1. For a v2i1
8273 // we convert to a v4i1 compare to fill in the two halves of the i64 as i32s.
8274 if (VT == MVT::v2i1) {
8275 SDValue BC = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, Shuffled);
8276 SDValue Cmp = DAG.getNode(ARMISD::VCMPZ, dl, MVT::v4i1, BC,
8277 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8278 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v2i1, Cmp);
8279 }
8280 return DAG.getNode(ARMISD::VCMPZ, dl, VT, Shuffled,
8281 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8282}
8283
8285 ArrayRef<int> ShuffleMask,
8286 SelectionDAG &DAG) {
8287 // Attempt to lower the vector shuffle using as many whole register movs as
8288 // possible. This is useful for types smaller than 32bits, which would
8289 // often otherwise become a series for grp movs.
8290 SDLoc dl(Op);
8291 EVT VT = Op.getValueType();
8292 if (VT.getScalarSizeInBits() >= 32)
8293 return SDValue();
8294
8295 assert((VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v16i8) &&
8296 "Unexpected vector type");
8297 int NumElts = VT.getVectorNumElements();
8298 int QuarterSize = NumElts / 4;
8299 // The four final parts of the vector, as i32's
8300 SDValue Parts[4];
8301
8302 // Look for full lane vmovs like <0,1,2,3> or <u,5,6,7> etc, (but not
8303 // <u,u,u,u>), returning the vmov lane index
8304 auto getMovIdx = [](ArrayRef<int> ShuffleMask, int Start, int Length) {
8305 // Detect which mov lane this would be from the first non-undef element.
8306 int MovIdx = -1;
8307 for (int i = 0; i < Length; i++) {
8308 if (ShuffleMask[Start + i] >= 0) {
8309 if (ShuffleMask[Start + i] % Length != i)
8310 return -1;
8311 MovIdx = ShuffleMask[Start + i] / Length;
8312 break;
8313 }
8314 }
8315 // If all items are undef, leave this for other combines
8316 if (MovIdx == -1)
8317 return -1;
8318 // Check the remaining values are the correct part of the same mov
8319 for (int i = 1; i < Length; i++) {
8320 if (ShuffleMask[Start + i] >= 0 &&
8321 (ShuffleMask[Start + i] / Length != MovIdx ||
8322 ShuffleMask[Start + i] % Length != i))
8323 return -1;
8324 }
8325 return MovIdx;
8326 };
8327
8328 for (int Part = 0; Part < 4; ++Part) {
8329 // Does this part look like a mov
8330 int Elt = getMovIdx(ShuffleMask, Part * QuarterSize, QuarterSize);
8331 if (Elt != -1) {
8332 SDValue Input = Op->getOperand(0);
8333 if (Elt >= 4) {
8334 Input = Op->getOperand(1);
8335 Elt -= 4;
8336 }
8337 SDValue BitCast = DAG.getBitcast(MVT::v4f32, Input);
8338 Parts[Part] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f32, BitCast,
8339 DAG.getConstant(Elt, dl, MVT::i32));
8340 }
8341 }
8342
8343 // Nothing interesting found, just return
8344 if (!Parts[0] && !Parts[1] && !Parts[2] && !Parts[3])
8345 return SDValue();
8346
8347 // The other parts need to be built with the old shuffle vector, cast to a
8348 // v4i32 and extract_vector_elts
8349 if (!Parts[0] || !Parts[1] || !Parts[2] || !Parts[3]) {
8350 SmallVector<int, 16> NewShuffleMask;
8351 for (int Part = 0; Part < 4; ++Part)
8352 for (int i = 0; i < QuarterSize; i++)
8353 NewShuffleMask.push_back(
8354 Parts[Part] ? -1 : ShuffleMask[Part * QuarterSize + i]);
8355 SDValue NewShuffle = DAG.getVectorShuffle(
8356 VT, dl, Op->getOperand(0), Op->getOperand(1), NewShuffleMask);
8357 SDValue BitCast = DAG.getBitcast(MVT::v4f32, NewShuffle);
8358
8359 for (int Part = 0; Part < 4; ++Part)
8360 if (!Parts[Part])
8361 Parts[Part] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f32,
8362 BitCast, DAG.getConstant(Part, dl, MVT::i32));
8363 }
8364 // Build a vector out of the various parts and bitcast it back to the original
8365 // type.
8366 SDValue NewVec = DAG.getNode(ARMISD::BUILD_VECTOR, dl, MVT::v4f32, Parts);
8367 return DAG.getBitcast(VT, NewVec);
8368}
8369
8371 ArrayRef<int> ShuffleMask,
8372 SelectionDAG &DAG) {
8373 SDValue V1 = Op.getOperand(0);
8374 SDValue V2 = Op.getOperand(1);
8375 EVT VT = Op.getValueType();
8376 unsigned NumElts = VT.getVectorNumElements();
8377
8378 // An One-Off Identity mask is one that is mostly an identity mask from as
8379 // single source but contains a single element out-of-place, either from a
8380 // different vector or from another position in the same vector. As opposed to
8381 // lowering this via a ARMISD::BUILD_VECTOR we can generate an extract/insert
8382 // pair directly.
8383 auto isOneOffIdentityMask = [](ArrayRef<int> Mask, EVT VT, int BaseOffset,
8384 int &OffElement) {
8385 OffElement = -1;
8386 int NonUndef = 0;
8387 for (int i = 0, NumMaskElts = Mask.size(); i < NumMaskElts; ++i) {
8388 if (Mask[i] == -1)
8389 continue;
8390 NonUndef++;
8391 if (Mask[i] != i + BaseOffset) {
8392 if (OffElement == -1)
8393 OffElement = i;
8394 else
8395 return false;
8396 }
8397 }
8398 return NonUndef > 2 && OffElement != -1;
8399 };
8400 int OffElement;
8401 SDValue VInput;
8402 if (isOneOffIdentityMask(ShuffleMask, VT, 0, OffElement))
8403 VInput = V1;
8404 else if (isOneOffIdentityMask(ShuffleMask, VT, NumElts, OffElement))
8405 VInput = V2;
8406 else
8407 return SDValue();
8408
8409 SDLoc dl(Op);
8410 EVT SVT = VT.getScalarType() == MVT::i8 || VT.getScalarType() == MVT::i16
8411 ? MVT::i32
8412 : VT.getScalarType();
8413 SDValue Elt = DAG.getNode(
8414 ISD::EXTRACT_VECTOR_ELT, dl, SVT,
8415 ShuffleMask[OffElement] < (int)NumElts ? V1 : V2,
8416 DAG.getVectorIdxConstant(ShuffleMask[OffElement] % NumElts, dl));
8417 return DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VT, VInput, Elt,
8418 DAG.getVectorIdxConstant(OffElement % NumElts, dl));
8419}
8420
8422 const ARMSubtarget *ST) {
8423 SDValue V1 = Op.getOperand(0);
8424 SDValue V2 = Op.getOperand(1);
8425 SDLoc dl(Op);
8426 EVT VT = Op.getValueType();
8428 unsigned EltSize = VT.getScalarSizeInBits();
8429
8430 if (ST->hasMVEIntegerOps() && EltSize == 1)
8431 return LowerVECTOR_SHUFFLE_i1(Op, DAG, ST);
8432
8433 // Convert shuffles that are directly supported on NEON to target-specific
8434 // DAG nodes, instead of keeping them as shuffles and matching them again
8435 // during code selection. This is more efficient and avoids the possibility
8436 // of inconsistencies between legalization and selection.
8437 // FIXME: floating-point vectors should be canonicalized to integer vectors
8438 // of the same time so that they get CSEd properly.
8439 ArrayRef<int> ShuffleMask = SVN->getMask();
8440
8441 if (EltSize <= 32) {
8442 if (SVN->isSplat()) {
8443 int Lane = SVN->getSplatIndex();
8444 // If this is undef splat, generate it via "just" vdup, if possible.
8445 if (Lane == -1) Lane = 0;
8446
8447 // Test if V1 is a SCALAR_TO_VECTOR.
8448 if (Lane == 0 && V1.getOpcode() == ISD::SCALAR_TO_VECTOR) {
8449 return DAG.getNode(ARMISD::VDUP, dl, VT, V1.getOperand(0));
8450 }
8451 // Test if V1 is a BUILD_VECTOR which is equivalent to a SCALAR_TO_VECTOR
8452 // (and probably will turn into a SCALAR_TO_VECTOR once legalization
8453 // reaches it).
8454 if (Lane == 0 && V1.getOpcode() == ISD::BUILD_VECTOR &&
8455 !isa<ConstantSDNode>(V1.getOperand(0))) {
8456 bool IsScalarToVector = true;
8457 for (unsigned i = 1, e = V1.getNumOperands(); i != e; ++i)
8458 if (!V1.getOperand(i).isUndef()) {
8459 IsScalarToVector = false;
8460 break;
8461 }
8462 if (IsScalarToVector)
8463 return DAG.getNode(ARMISD::VDUP, dl, VT, V1.getOperand(0));
8464 }
8465 return DAG.getNode(ARMISD::VDUPLANE, dl, VT, V1,
8466 DAG.getConstant(Lane, dl, MVT::i32));
8467 }
8468
8469 bool ReverseVEXT = false;
8470 unsigned Imm = 0;
8471 if (ST->hasNEON() && isVEXTMask(ShuffleMask, VT, ReverseVEXT, Imm)) {
8472 if (ReverseVEXT)
8473 std::swap(V1, V2);
8474 return DAG.getNode(ARMISD::VEXT, dl, VT, V1, V2,
8475 DAG.getConstant(Imm, dl, MVT::i32));
8476 }
8477
8478 if (isVREVMask(ShuffleMask, VT, 64))
8479 return DAG.getNode(ARMISD::VREV64, dl, VT, V1);
8480 if (isVREVMask(ShuffleMask, VT, 32))
8481 return DAG.getNode(ARMISD::VREV32, dl, VT, V1);
8482 if (isVREVMask(ShuffleMask, VT, 16))
8483 return DAG.getNode(ARMISD::VREV16, dl, VT, V1);
8484
8485 if (ST->hasNEON() && V2->isUndef() && isSingletonVEXTMask(ShuffleMask, VT, Imm)) {
8486 return DAG.getNode(ARMISD::VEXT, dl, VT, V1, V1,
8487 DAG.getConstant(Imm, dl, MVT::i32));
8488 }
8489
8490 // Check for Neon shuffles that modify both input vectors in place.
8491 // If both results are used, i.e., if there are two shuffles with the same
8492 // source operands and with masks corresponding to both results of one of
8493 // these operations, DAG memoization will ensure that a single node is
8494 // used for both shuffles.
8495 unsigned WhichResult = 0;
8496 bool isV_UNDEF = false;
8497 if (ST->hasNEON()) {
8498 if (unsigned ShuffleOpc = isNEONTwoResultShuffleMask(
8499 ShuffleMask, VT, WhichResult, isV_UNDEF)) {
8500 if (isV_UNDEF)
8501 V2 = V1;
8502 return DAG.getNode(ShuffleOpc, dl, DAG.getVTList(VT, VT), V1, V2)
8503 .getValue(WhichResult);
8504 }
8505 }
8506 if (ST->hasMVEIntegerOps()) {
8507 if (isVMOVNMask(ShuffleMask, VT, false, false))
8508 return DAG.getNode(ARMISD::VMOVN, dl, VT, V2, V1,
8509 DAG.getConstant(0, dl, MVT::i32));
8510 if (isVMOVNMask(ShuffleMask, VT, true, false))
8511 return DAG.getNode(ARMISD::VMOVN, dl, VT, V1, V2,
8512 DAG.getConstant(1, dl, MVT::i32));
8513 if (isVMOVNMask(ShuffleMask, VT, true, true))
8514 return DAG.getNode(ARMISD::VMOVN, dl, VT, V1, V1,
8515 DAG.getConstant(1, dl, MVT::i32));
8516 }
8517
8518 // Also check for these shuffles through CONCAT_VECTORS: we canonicalize
8519 // shuffles that produce a result larger than their operands with:
8520 // shuffle(concat(v1, undef), concat(v2, undef))
8521 // ->
8522 // shuffle(concat(v1, v2), undef)
8523 // because we can access quad vectors (see PerformVECTOR_SHUFFLECombine).
8524 //
8525 // This is useful in the general case, but there are special cases where
8526 // native shuffles produce larger results: the two-result ops.
8527 //
8528 // Look through the concat when lowering them:
8529 // shuffle(concat(v1, v2), undef)
8530 // ->
8531 // concat(VZIP(v1, v2):0, :1)
8532 //
8533 if (ST->hasNEON() && V1->getOpcode() == ISD::CONCAT_VECTORS && V2->isUndef()) {
8534 SDValue SubV1 = V1->getOperand(0);
8535 SDValue SubV2 = V1->getOperand(1);
8536 EVT SubVT = SubV1.getValueType();
8537
8538 // We expect these to have been canonicalized to -1.
8539 assert(llvm::all_of(ShuffleMask, [&](int i) {
8540 return i < (int)VT.getVectorNumElements();
8541 }) && "Unexpected shuffle index into UNDEF operand!");
8542
8543 if (unsigned ShuffleOpc = isNEONTwoResultShuffleMask(
8544 ShuffleMask, SubVT, WhichResult, isV_UNDEF)) {
8545 if (isV_UNDEF)
8546 SubV2 = SubV1;
8547 assert((WhichResult == 0) &&
8548 "In-place shuffle of concat can only have one result!");
8549 SDValue Res = DAG.getNode(ShuffleOpc, dl, DAG.getVTList(SubVT, SubVT),
8550 SubV1, SubV2);
8551 return DAG.getNode(ISD::CONCAT_VECTORS, dl, VT, Res.getValue(0),
8552 Res.getValue(1));
8553 }
8554 }
8555 }
8556
8557 if (ST->hasMVEIntegerOps() && EltSize <= 32 &&
8558 (ST->hasFullFP16() || VT != MVT::v8f16)) {
8559 if (SDValue V = LowerVECTOR_SHUFFLEUsingOneOff(Op, ShuffleMask, DAG))
8560 return V;
8561
8562 for (bool Top : {false, true}) {
8563 for (bool SingleSource : {false, true}) {
8564 if (isTruncMask(ShuffleMask, VT, Top, SingleSource)) {
8565 MVT FromSVT = MVT::getIntegerVT(EltSize * 2);
8566 MVT FromVT = MVT::getVectorVT(FromSVT, ShuffleMask.size() / 2);
8567 SDValue Lo = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, FromVT, V1);
8568 SDValue Hi = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, FromVT,
8569 SingleSource ? V1 : V2);
8570 if (Top) {
8571 SDValue Amt = DAG.getConstant(EltSize, dl, FromVT);
8572 Lo = DAG.getNode(ISD::SRL, dl, FromVT, Lo, Amt);
8573 Hi = DAG.getNode(ISD::SRL, dl, FromVT, Hi, Amt);
8574 }
8575 return DAG.getNode(ARMISD::MVETRUNC, dl, VT, Lo, Hi);
8576 }
8577 }
8578 }
8579 }
8580
8581 // If the shuffle is not directly supported and it has 4 elements, use
8582 // the PerfectShuffle-generated table to synthesize it from other shuffles.
8583 unsigned NumElts = VT.getVectorNumElements();
8584 if (NumElts == 4) {
8585 unsigned PFIndexes[4];
8586 for (unsigned i = 0; i != 4; ++i) {
8587 if (ShuffleMask[i] < 0)
8588 PFIndexes[i] = 8;
8589 else
8590 PFIndexes[i] = ShuffleMask[i];
8591 }
8592
8593 // Compute the index in the perfect shuffle table.
8594 unsigned PFTableIndex =
8595 PFIndexes[0]*9*9*9+PFIndexes[1]*9*9+PFIndexes[2]*9+PFIndexes[3];
8596 unsigned PFEntry = PerfectShuffleTable[PFTableIndex];
8597 unsigned Cost = (PFEntry >> 30);
8598
8599 if (Cost <= 4) {
8600 if (ST->hasNEON())
8601 return GeneratePerfectShuffle(PFEntry, V1, V2, DAG, dl);
8602 else if (isLegalMVEShuffleOp(PFEntry)) {
8603 unsigned LHSID = (PFEntry >> 13) & ((1 << 13)-1);
8604 unsigned RHSID = (PFEntry >> 0) & ((1 << 13)-1);
8605 unsigned PFEntryLHS = PerfectShuffleTable[LHSID];
8606 unsigned PFEntryRHS = PerfectShuffleTable[RHSID];
8607 if (isLegalMVEShuffleOp(PFEntryLHS) && isLegalMVEShuffleOp(PFEntryRHS))
8608 return GeneratePerfectShuffle(PFEntry, V1, V2, DAG, dl);
8609 }
8610 }
8611 }
8612
8613 // Implement shuffles with 32- or 64-bit elements as ARMISD::BUILD_VECTORs.
8614 if (EltSize >= 32) {
8615 // Do the expansion with floating-point types, since that is what the VFP
8616 // registers are defined to use, and since i64 is not legal.
8617 EVT EltVT = EVT::getFloatingPointVT(EltSize);
8618 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts);
8619 V1 = DAG.getNode(ISD::BITCAST, dl, VecVT, V1);
8620 V2 = DAG.getNode(ISD::BITCAST, dl, VecVT, V2);
8622 for (unsigned i = 0; i < NumElts; ++i) {
8623 if (ShuffleMask[i] < 0)
8624 Ops.push_back(DAG.getUNDEF(EltVT));
8625 else
8626 Ops.push_back(DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT,
8627 ShuffleMask[i] < (int)NumElts ? V1 : V2,
8628 DAG.getConstant(ShuffleMask[i] & (NumElts-1),
8629 dl, MVT::i32)));
8630 }
8631 SDValue Val = DAG.getNode(ARMISD::BUILD_VECTOR, dl, VecVT, Ops);
8632 return DAG.getNode(ISD::BITCAST, dl, VT, Val);
8633 }
8634
8635 if ((VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v16i8) &&
8636 isReverseMask(ShuffleMask, VT))
8637 return LowerReverse_VECTOR_SHUFFLE(Op, DAG);
8638
8639 if (ST->hasNEON() && VT == MVT::v8i8)
8640 if (SDValue NewOp = LowerVECTOR_SHUFFLEv8i8(Op, ShuffleMask, DAG))
8641 return NewOp;
8642
8643 if (ST->hasMVEIntegerOps())
8644 if (SDValue NewOp = LowerVECTOR_SHUFFLEUsingMovs(Op, ShuffleMask, DAG))
8645 return NewOp;
8646
8647 // Lower v8f16 via v8i16 to avoid invalid f16 nodes.
8648 if (VT == MVT::v8f16 && !ST->hasFullFP16()) {
8649 SDValue BC0 =
8650 DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v8i16, Op.getOperand(0));
8651 SDValue BC1 =
8652 DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v8i16, Op.getOperand(1));
8653 SDValue Shuf = DAG.getVectorShuffle(MVT::v8i16, dl, BC0, BC1, ShuffleMask);
8654 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Shuf);
8655 }
8656
8657 return SDValue();
8658}
8659
8661 const ARMSubtarget *ST) {
8662 EVT VecVT = Op.getOperand(0).getValueType();
8663 SDLoc dl(Op);
8664
8665 assert(ST->hasMVEIntegerOps() &&
8666 "LowerINSERT_VECTOR_ELT_i1 called without MVE!");
8667
8668 SDValue Conv =
8669 DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Op->getOperand(0));
8670 unsigned Lane = Op.getConstantOperandVal(2);
8671 unsigned LaneWidth =
8673 unsigned Mask = ((1 << LaneWidth) - 1) << Lane * LaneWidth;
8674 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND_INREG, dl, MVT::i32,
8675 Op.getOperand(1), DAG.getValueType(MVT::i1));
8676 SDValue BFI = DAG.getNode(ARMISD::BFI, dl, MVT::i32, Conv, Ext,
8677 DAG.getConstant(~Mask, dl, MVT::i32));
8678 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, Op.getValueType(), BFI);
8679}
8680
8681SDValue ARMTargetLowering::LowerINSERT_VECTOR_ELT(SDValue Op,
8682 SelectionDAG &DAG) const {
8683 // INSERT_VECTOR_ELT is legal only for immediate indexes.
8684 SDValue Lane = Op.getOperand(2);
8685 if (!isa<ConstantSDNode>(Lane))
8686 return SDValue();
8687
8688 SDValue Elt = Op.getOperand(1);
8689 EVT EltVT = Elt.getValueType();
8690
8691 if (Subtarget->hasMVEIntegerOps() &&
8692 Op.getValueType().getScalarSizeInBits() == 1)
8693 return LowerINSERT_VECTOR_ELT_i1(Op, DAG, Subtarget);
8694
8695 if (getTypeAction(*DAG.getContext(), EltVT) ==
8697 // INSERT_VECTOR_ELT doesn't want f16 operands promoting to f32,
8698 // but the type system will try to do that if we don't intervene.
8699 // Reinterpret any such vector-element insertion as one with the
8700 // corresponding integer types.
8701
8702 SDLoc dl(Op);
8703
8704 EVT IEltVT = MVT::getIntegerVT(EltVT.getScalarSizeInBits());
8705 assert(getTypeAction(*DAG.getContext(), IEltVT) !=
8707
8708 SDValue VecIn = Op.getOperand(0);
8709 EVT VecVT = VecIn.getValueType();
8710 EVT IVecVT = EVT::getVectorVT(*DAG.getContext(), IEltVT,
8711 VecVT.getVectorNumElements());
8712
8713 SDValue IElt = DAG.getNode(ISD::BITCAST, dl, IEltVT, Elt);
8714 SDValue IVecIn = DAG.getNode(ISD::BITCAST, dl, IVecVT, VecIn);
8715 SDValue IVecOut = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, IVecVT,
8716 IVecIn, IElt, Lane);
8717 return DAG.getNode(ISD::BITCAST, dl, VecVT, IVecOut);
8718 }
8719
8720 return Op;
8721}
8722
8724 const ARMSubtarget *ST) {
8725 EVT VecVT = Op.getOperand(0).getValueType();
8726 SDLoc dl(Op);
8727
8728 assert(ST->hasMVEIntegerOps() &&
8729 "LowerINSERT_VECTOR_ELT_i1 called without MVE!");
8730
8731 SDValue Conv =
8732 DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Op->getOperand(0));
8733 unsigned Lane = Op.getConstantOperandVal(1);
8734 unsigned LaneWidth =
8736 SDValue Shift = DAG.getNode(ISD::SRL, dl, MVT::i32, Conv,
8737 DAG.getConstant(Lane * LaneWidth, dl, MVT::i32));
8738 return Shift;
8739}
8740
8742 const ARMSubtarget *ST) {
8743 // EXTRACT_VECTOR_ELT is legal only for immediate indexes.
8744 SDValue Lane = Op.getOperand(1);
8745 if (!isa<ConstantSDNode>(Lane))
8746 return SDValue();
8747
8748 SDValue Vec = Op.getOperand(0);
8749 EVT VT = Vec.getValueType();
8750
8751 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1)
8752 return LowerEXTRACT_VECTOR_ELT_i1(Op, DAG, ST);
8753
8754 if (Op.getValueType() == MVT::i32 && Vec.getScalarValueSizeInBits() < 32) {
8755 SDLoc dl(Op);
8756 return DAG.getNode(ARMISD::VGETLANEu, dl, MVT::i32, Vec, Lane);
8757 }
8758
8759 return Op;
8760}
8761
8763 const ARMSubtarget *ST) {
8764 SDLoc dl(Op);
8765 assert(Op.getValueType().getScalarSizeInBits() == 1 &&
8766 "Unexpected custom CONCAT_VECTORS lowering");
8767 assert(isPowerOf2_32(Op.getNumOperands()) &&
8768 "Unexpected custom CONCAT_VECTORS lowering");
8769 assert(ST->hasMVEIntegerOps() &&
8770 "CONCAT_VECTORS lowering only supported for MVE");
8771
8772 auto ConcatPair = [&](SDValue V1, SDValue V2) {
8773 EVT Op1VT = V1.getValueType();
8774 EVT Op2VT = V2.getValueType();
8775 assert(Op1VT == Op2VT && "Operand types don't match!");
8776 assert((Op1VT == MVT::v2i1 || Op1VT == MVT::v4i1 || Op1VT == MVT::v8i1) &&
8777 "Unexpected i1 concat operations!");
8778 EVT VT = Op1VT.getDoubleNumVectorElementsVT(*DAG.getContext());
8779
8780 SDValue NewV1 = PromoteMVEPredVector(dl, V1, Op1VT, DAG);
8781 SDValue NewV2 = PromoteMVEPredVector(dl, V2, Op2VT, DAG);
8782
8783 // We now have Op1 + Op2 promoted to vectors of integers, where v8i1 gets
8784 // promoted to v8i16, etc.
8785 MVT ElType =
8787 unsigned NumElts = 2 * Op1VT.getVectorNumElements();
8788
8789 EVT ConcatVT = MVT::getVectorVT(ElType, NumElts);
8790 if (Op1VT == MVT::v4i1 || Op1VT == MVT::v8i1) {
8791 // Use MVETRUNC to truncate the combined NewV1::NewV2 into the smaller
8792 // ConcatVT.
8793 SDValue ConVec =
8794 DAG.getNode(ARMISD::MVETRUNC, dl, ConcatVT, NewV1, NewV2);
8795 return DAG.getNode(ARMISD::VCMPZ, dl, VT, ConVec,
8796 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8797 }
8798
8799 // Extract the vector elements from Op1 and Op2 one by one and truncate them
8800 // to be the right size for the destination. For example, if Op1 is v4i1
8801 // then the promoted vector is v4i32. The result of concatenation gives a
8802 // v8i1, which when promoted is v8i16. That means each i32 element from Op1
8803 // needs truncating to i16 and inserting in the result.
8804 auto ExtractInto = [&DAG, &dl](SDValue NewV, SDValue ConVec, unsigned &j) {
8805 EVT NewVT = NewV.getValueType();
8806 EVT ConcatVT = ConVec.getValueType();
8807 unsigned ExtScale = 1;
8808 if (NewVT == MVT::v2f64) {
8809 NewV = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, NewV);
8810 ExtScale = 2;
8811 }
8812 for (unsigned i = 0, e = NewVT.getVectorNumElements(); i < e; i++, j++) {
8813 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, NewV,
8814 DAG.getIntPtrConstant(i * ExtScale, dl));
8815 ConVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, ConcatVT, ConVec, Elt,
8816 DAG.getConstant(j, dl, MVT::i32));
8817 }
8818 return ConVec;
8819 };
8820 unsigned j = 0;
8821 SDValue ConVec = DAG.getNode(ISD::UNDEF, dl, ConcatVT);
8822 ConVec = ExtractInto(NewV1, ConVec, j);
8823 ConVec = ExtractInto(NewV2, ConVec, j);
8824
8825 // Now return the result of comparing the subvector with zero, which will
8826 // generate a real predicate, i.e. v4i1, v8i1 or v16i1.
8827 return DAG.getNode(ARMISD::VCMPZ, dl, VT, ConVec,
8828 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8829 };
8830
8831 // Concat each pair of subvectors and pack into the lower half of the array.
8832 SmallVector<SDValue> ConcatOps(Op->ops());
8833 while (ConcatOps.size() > 1) {
8834 for (unsigned I = 0, E = ConcatOps.size(); I != E; I += 2) {
8835 SDValue V1 = ConcatOps[I];
8836 SDValue V2 = ConcatOps[I + 1];
8837 ConcatOps[I / 2] = ConcatPair(V1, V2);
8838 }
8839 ConcatOps.resize(ConcatOps.size() / 2);
8840 }
8841 return ConcatOps[0];
8842}
8843
8845 const ARMSubtarget *ST) {
8846 EVT VT = Op->getValueType(0);
8847 if (ST->hasMVEIntegerOps() && VT.getScalarSizeInBits() == 1)
8848 return LowerCONCAT_VECTORS_i1(Op, DAG, ST);
8849
8850 // The only time a CONCAT_VECTORS operation can have legal types is when
8851 // two 64-bit vectors are concatenated to a 128-bit vector.
8852 assert(Op.getValueType().is128BitVector() && Op.getNumOperands() == 2 &&
8853 "unexpected CONCAT_VECTORS");
8854 SDLoc dl(Op);
8855 SDValue Val = DAG.getUNDEF(MVT::v2f64);
8856 SDValue Op0 = Op.getOperand(0);
8857 SDValue Op1 = Op.getOperand(1);
8858 if (!Op0.isUndef())
8859 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Val,
8860 DAG.getNode(ISD::BITCAST, dl, MVT::f64, Op0),
8861 DAG.getIntPtrConstant(0, dl));
8862 if (!Op1.isUndef())
8863 Val = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, MVT::v2f64, Val,
8864 DAG.getNode(ISD::BITCAST, dl, MVT::f64, Op1),
8865 DAG.getIntPtrConstant(1, dl));
8866 return DAG.getNode(ISD::BITCAST, dl, Op.getValueType(), Val);
8867}
8868
8870 const ARMSubtarget *ST) {
8871 SDValue V1 = Op.getOperand(0);
8872 SDValue V2 = Op.getOperand(1);
8873 SDLoc dl(Op);
8874 EVT VT = Op.getValueType();
8875 EVT Op1VT = V1.getValueType();
8876 unsigned NumElts = VT.getVectorNumElements();
8877 unsigned Index = V2->getAsZExtVal();
8878
8879 assert(VT.getScalarSizeInBits() == 1 &&
8880 "Unexpected custom EXTRACT_SUBVECTOR lowering");
8881 assert(ST->hasMVEIntegerOps() &&
8882 "EXTRACT_SUBVECTOR lowering only supported for MVE");
8883
8884 SDValue NewV1 = PromoteMVEPredVector(dl, V1, Op1VT, DAG);
8885
8886 // We now have Op1 promoted to a vector of integers, where v8i1 gets
8887 // promoted to v8i16, etc.
8888
8890
8891 if (NumElts == 2) {
8892 EVT SubVT = MVT::v4i32;
8893 SDValue SubVec = DAG.getNode(ISD::UNDEF, dl, SubVT);
8894 for (unsigned i = Index, j = 0; i < (Index + NumElts); i++, j += 2) {
8895 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, NewV1,
8896 DAG.getIntPtrConstant(i, dl));
8897 SubVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, SubVT, SubVec, Elt,
8898 DAG.getConstant(j, dl, MVT::i32));
8899 SubVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, SubVT, SubVec, Elt,
8900 DAG.getConstant(j + 1, dl, MVT::i32));
8901 }
8902 SDValue Cmp = DAG.getNode(ARMISD::VCMPZ, dl, MVT::v4i1, SubVec,
8903 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8904 return DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v2i1, Cmp);
8905 }
8906
8907 EVT SubVT = MVT::getVectorVT(ElType, NumElts);
8908 SDValue SubVec = DAG.getNode(ISD::UNDEF, dl, SubVT);
8909 for (unsigned i = Index, j = 0; i < (Index + NumElts); i++, j++) {
8910 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, NewV1,
8911 DAG.getIntPtrConstant(i, dl));
8912 SubVec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, SubVT, SubVec, Elt,
8913 DAG.getConstant(j, dl, MVT::i32));
8914 }
8915
8916 // Now return the result of comparing the subvector with zero,
8917 // which will generate a real predicate, i.e. v4i1, v8i1 or v16i1.
8918 return DAG.getNode(ARMISD::VCMPZ, dl, VT, SubVec,
8919 DAG.getConstant(ARMCC::NE, dl, MVT::i32));
8920}
8921
8922// Turn a truncate into a predicate (an i1 vector) into icmp(and(x, 1), 0).
8924 const ARMSubtarget *ST) {
8925 assert(ST->hasMVEIntegerOps() && "Expected MVE!");
8926 EVT VT = N->getValueType(0);
8927 assert((VT == MVT::v16i1 || VT == MVT::v8i1 || VT == MVT::v4i1) &&
8928 "Expected a vector i1 type!");
8929 SDValue Op = N->getOperand(0);
8930 EVT FromVT = Op.getValueType();
8931 SDLoc DL(N);
8932
8933 SDValue And =
8934 DAG.getNode(ISD::AND, DL, FromVT, Op, DAG.getConstant(1, DL, FromVT));
8935 return DAG.getNode(ISD::SETCC, DL, VT, And, DAG.getConstant(0, DL, FromVT),
8936 DAG.getCondCode(ISD::SETNE));
8937}
8938
8940 const ARMSubtarget *Subtarget) {
8941 if (!Subtarget->hasMVEIntegerOps())
8942 return SDValue();
8943
8944 EVT ToVT = N->getValueType(0);
8945 if (ToVT.getScalarType() == MVT::i1)
8946 return LowerTruncatei1(N, DAG, Subtarget);
8947
8948 // MVE does not have a single instruction to perform the truncation of a v4i32
8949 // into the lower half of a v8i16, in the same way that a NEON vmovn would.
8950 // Most of the instructions in MVE follow the 'Beats' system, where moving
8951 // values from different lanes is usually something that the instructions
8952 // avoid.
8953 //
8954 // Instead it has top/bottom instructions such as VMOVLT/B and VMOVNT/B,
8955 // which take a the top/bottom half of a larger lane and extend it (or do the
8956 // opposite, truncating into the top/bottom lane from a larger lane). Note
8957 // that because of the way we widen lanes, a v4i16 is really a v4i32 using the
8958 // bottom 16bits from each vector lane. This works really well with T/B
8959 // instructions, but that doesn't extend to v8i32->v8i16 where the lanes need
8960 // to move order.
8961 //
8962 // But truncates and sext/zext are always going to be fairly common from llvm.
8963 // We have several options for how to deal with them:
8964 // - Wherever possible combine them into an instruction that makes them
8965 // "free". This includes loads/stores, which can perform the trunc as part
8966 // of the memory operation. Or certain shuffles that can be turned into
8967 // VMOVN/VMOVL.
8968 // - Lane Interleaving to transform blocks surrounded by ext/trunc. So
8969 // trunc(mul(sext(a), sext(b))) may become
8970 // VMOVNT(VMUL(VMOVLB(a), VMOVLB(b)), VMUL(VMOVLT(a), VMOVLT(b))). (Which in
8971 // this case can use VMULL). This is performed in the
8972 // MVELaneInterleavingPass.
8973 // - Otherwise we have an option. By default we would expand the
8974 // zext/sext/trunc into a series of lane extract/inserts going via GPR
8975 // registers. One for each vector lane in the vector. This can obviously be
8976 // very expensive.
8977 // - The other option is to use the fact that loads/store can extend/truncate
8978 // to turn a trunc into two truncating stack stores and a stack reload. This
8979 // becomes 3 back-to-back memory operations, but at least that is less than
8980 // all the insert/extracts.
8981 //
8982 // In order to do the last, we convert certain trunc's into MVETRUNC, which
8983 // are either optimized where they can be, or eventually lowered into stack
8984 // stores/loads. This prevents us from splitting a v8i16 trunc into two stores
8985 // two early, where other instructions would be better, and stops us from
8986 // having to reconstruct multiple buildvector shuffles into loads/stores.
8987 if (ToVT != MVT::v8i16 && ToVT != MVT::v16i8)
8988 return SDValue();
8989 EVT FromVT = N->getOperand(0).getValueType();
8990 if (FromVT != MVT::v8i32 && FromVT != MVT::v16i16)
8991 return SDValue();
8992
8993 SDValue Lo, Hi;
8994 std::tie(Lo, Hi) = DAG.SplitVectorOperand(N, 0);
8995 SDLoc DL(N);
8996 return DAG.getNode(ARMISD::MVETRUNC, DL, ToVT, Lo, Hi);
8997}
8998
9000 const ARMSubtarget *Subtarget) {
9001 if (!Subtarget->hasMVEIntegerOps())
9002 return SDValue();
9003
9004 // See LowerTruncate above for an explanation of MVEEXT/MVETRUNC.
9005
9006 EVT ToVT = N->getValueType(0);
9007 if (ToVT != MVT::v16i32 && ToVT != MVT::v8i32 && ToVT != MVT::v16i16)
9008 return SDValue();
9009 SDValue Op = N->getOperand(0);
9010 EVT FromVT = Op.getValueType();
9011 if (FromVT != MVT::v8i16 && FromVT != MVT::v16i8)
9012 return SDValue();
9013
9014 SDLoc DL(N);
9015 EVT ExtVT = ToVT.getHalfNumVectorElementsVT(*DAG.getContext());
9016 if (ToVT.getScalarType() == MVT::i32 && FromVT.getScalarType() == MVT::i8)
9017 ExtVT = MVT::v8i16;
9018
9019 unsigned Opcode =
9021 SDValue Ext = DAG.getNode(Opcode, DL, DAG.getVTList(ExtVT, ExtVT), Op);
9022 SDValue Ext1 = Ext.getValue(1);
9023
9024 if (ToVT.getScalarType() == MVT::i32 && FromVT.getScalarType() == MVT::i8) {
9025 Ext = DAG.getNode(N->getOpcode(), DL, MVT::v8i32, Ext);
9026 Ext1 = DAG.getNode(N->getOpcode(), DL, MVT::v8i32, Ext1);
9027 }
9028
9029 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ToVT, Ext, Ext1);
9030}
9031
9032/// isExtendedBUILD_VECTOR - Check if N is a constant BUILD_VECTOR where each
9033/// element has been zero/sign-extended, depending on the isSigned parameter,
9034/// from an integer type half its size.
9036 bool isSigned) {
9037 // A v2i64 BUILD_VECTOR will have been legalized to a BITCAST from v4i32.
9038 EVT VT = N->getValueType(0);
9039 if (VT == MVT::v2i64 && N->getOpcode() == ISD::BITCAST) {
9040 SDNode *BVN = N->getOperand(0).getNode();
9041 if (BVN->getValueType(0) != MVT::v4i32 ||
9042 BVN->getOpcode() != ISD::BUILD_VECTOR)
9043 return false;
9044 unsigned LoElt = DAG.getDataLayout().isBigEndian() ? 1 : 0;
9045 unsigned HiElt = 1 - LoElt;
9050 if (!Lo0 || !Hi0 || !Lo1 || !Hi1)
9051 return false;
9052 if (isSigned) {
9053 if (Hi0->getSExtValue() == Lo0->getSExtValue() >> 32 &&
9054 Hi1->getSExtValue() == Lo1->getSExtValue() >> 32)
9055 return true;
9056 } else {
9057 if (Hi0->isZero() && Hi1->isZero())
9058 return true;
9059 }
9060 return false;
9061 }
9062
9063 if (N->getOpcode() != ISD::BUILD_VECTOR)
9064 return false;
9065
9066 for (unsigned i = 0, e = N->getNumOperands(); i != e; ++i) {
9067 SDNode *Elt = N->getOperand(i).getNode();
9069 unsigned EltSize = VT.getScalarSizeInBits();
9070 unsigned HalfSize = EltSize / 2;
9071 if (isSigned) {
9072 if (!isIntN(HalfSize, C->getSExtValue()))
9073 return false;
9074 } else {
9075 if (!isUIntN(HalfSize, C->getZExtValue()))
9076 return false;
9077 }
9078 continue;
9079 }
9080 return false;
9081 }
9082
9083 return true;
9084}
9085
9086/// isSignExtended - Check if a node is a vector value that is sign-extended
9087/// or a constant BUILD_VECTOR with sign-extended elements.
9089 if (N->getOpcode() == ISD::SIGN_EXTEND || ISD::isSEXTLoad(N))
9090 return true;
9091 if (isExtendedBUILD_VECTOR(N, DAG, true))
9092 return true;
9093 return false;
9094}
9095
9096/// isZeroExtended - Check if a node is a vector value that is zero-extended (or
9097/// any-extended) or a constant BUILD_VECTOR with zero-extended elements.
9099 if (N->getOpcode() == ISD::ZERO_EXTEND || N->getOpcode() == ISD::ANY_EXTEND ||
9101 return true;
9102 if (isExtendedBUILD_VECTOR(N, DAG, false))
9103 return true;
9104 return false;
9105}
9106
9107static EVT getExtensionTo64Bits(const EVT &OrigVT) {
9108 if (OrigVT.getSizeInBits() >= 64)
9109 return OrigVT;
9110
9111 assert(OrigVT.isSimple() && "Expecting a simple value type");
9112
9113 MVT::SimpleValueType OrigSimpleTy = OrigVT.getSimpleVT().SimpleTy;
9114 switch (OrigSimpleTy) {
9115 default: llvm_unreachable("Unexpected Vector Type");
9116 case MVT::v2i8:
9117 case MVT::v2i16:
9118 return MVT::v2i32;
9119 case MVT::v4i8:
9120 return MVT::v4i16;
9121 }
9122}
9123
9124/// AddRequiredExtensionForVMULL - Add a sign/zero extension to extend the total
9125/// value size to 64 bits. We need a 64-bit D register as an operand to VMULL.
9126/// We insert the required extension here to get the vector to fill a D register.
9128 const EVT &OrigTy,
9129 const EVT &ExtTy,
9130 unsigned ExtOpcode) {
9131 // The vector originally had a size of OrigTy. It was then extended to ExtTy.
9132 // We expect the ExtTy to be 128-bits total. If the OrigTy is less than
9133 // 64-bits we need to insert a new extension so that it will be 64-bits.
9134 assert(ExtTy.is128BitVector() && "Unexpected extension size");
9135 if (OrigTy.getSizeInBits() >= 64)
9136 return N;
9137
9138 // Must extend size to at least 64 bits to be used as an operand for VMULL.
9139 EVT NewVT = getExtensionTo64Bits(OrigTy);
9140
9141 return DAG.getNode(ExtOpcode, SDLoc(N), NewVT, N);
9142}
9143
9144/// SkipLoadExtensionForVMULL - return a load of the original vector size that
9145/// does not do any sign/zero extension. If the original vector is less
9146/// than 64 bits, an appropriate extension will be added after the load to
9147/// reach a total size of 64 bits. We have to add the extension separately
9148/// because ARM does not have a sign/zero extending load for vectors.
9150 EVT ExtendedTy = getExtensionTo64Bits(LD->getMemoryVT());
9151
9152 // The load already has the right type.
9153 if (ExtendedTy == LD->getMemoryVT())
9154 return DAG.getLoad(LD->getMemoryVT(), SDLoc(LD), LD->getChain(),
9155 LD->getBasePtr(), LD->getPointerInfo(), LD->getAlign(),
9156 LD->getMemOperand()->getFlags());
9157
9158 // We need to create a zextload/sextload. We cannot just create a load
9159 // followed by a zext/zext node because LowerMUL is also run during normal
9160 // operation legalization where we can't create illegal types.
9161 return DAG.getExtLoad(LD->getExtensionType(), SDLoc(LD), ExtendedTy,
9162 LD->getChain(), LD->getBasePtr(), LD->getPointerInfo(),
9163 LD->getMemoryVT(), LD->getAlign(),
9164 LD->getMemOperand()->getFlags());
9165}
9166
9167/// SkipExtensionForVMULL - For a node that is a SIGN_EXTEND, ZERO_EXTEND,
9168/// ANY_EXTEND, extending load, or BUILD_VECTOR with extended elements, return
9169/// the unextended value. The unextended vector should be 64 bits so that it can
9170/// be used as an operand to a VMULL instruction. If the original vector size
9171/// before extension is less than 64 bits we add a an extension to resize
9172/// the vector to 64 bits.
9174 if (N->getOpcode() == ISD::SIGN_EXTEND ||
9175 N->getOpcode() == ISD::ZERO_EXTEND || N->getOpcode() == ISD::ANY_EXTEND)
9176 return AddRequiredExtensionForVMULL(N->getOperand(0), DAG,
9177 N->getOperand(0)->getValueType(0),
9178 N->getValueType(0),
9179 N->getOpcode());
9180
9181 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) {
9182 assert((ISD::isSEXTLoad(LD) || ISD::isZEXTLoad(LD)) &&
9183 "Expected extending load");
9184
9185 SDValue newLoad = SkipLoadExtensionForVMULL(LD, DAG);
9186 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), newLoad.getValue(1));
9187 unsigned Opcode = ISD::isSEXTLoad(LD) ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
9188 SDValue extLoad =
9189 DAG.getNode(Opcode, SDLoc(newLoad), LD->getValueType(0), newLoad);
9190 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 0), extLoad);
9191
9192 return newLoad;
9193 }
9194
9195 // Otherwise, the value must be a BUILD_VECTOR. For v2i64, it will
9196 // have been legalized as a BITCAST from v4i32.
9197 if (N->getOpcode() == ISD::BITCAST) {
9198 SDNode *BVN = N->getOperand(0).getNode();
9200 BVN->getValueType(0) == MVT::v4i32 && "expected v4i32 BUILD_VECTOR");
9201 unsigned LowElt = DAG.getDataLayout().isBigEndian() ? 1 : 0;
9202 return DAG.getBuildVector(
9203 MVT::v2i32, SDLoc(N),
9204 {BVN->getOperand(LowElt), BVN->getOperand(LowElt + 2)});
9205 }
9206 // Construct a new BUILD_VECTOR with elements truncated to half the size.
9207 assert(N->getOpcode() == ISD::BUILD_VECTOR && "expected BUILD_VECTOR");
9208 EVT VT = N->getValueType(0);
9209 unsigned EltSize = VT.getScalarSizeInBits() / 2;
9210 unsigned NumElts = VT.getVectorNumElements();
9211 MVT TruncVT = MVT::getIntegerVT(EltSize);
9213 SDLoc dl(N);
9214 for (unsigned i = 0; i != NumElts; ++i) {
9215 const APInt &CInt = N->getConstantOperandAPInt(i);
9216 // Element types smaller than 32 bits are not legal, so use i32 elements.
9217 // The values are implicitly truncated so sext vs. zext doesn't matter.
9218 Ops.push_back(DAG.getConstant(CInt.zextOrTrunc(32), dl, MVT::i32));
9219 }
9220 return DAG.getBuildVector(MVT::getVectorVT(TruncVT, NumElts), dl, Ops);
9221}
9222
9223static bool isAddSubSExt(SDNode *N, SelectionDAG &DAG) {
9224 unsigned Opcode = N->getOpcode();
9225 if (Opcode == ISD::ADD || Opcode == ISD::SUB) {
9226 SDNode *N0 = N->getOperand(0).getNode();
9227 SDNode *N1 = N->getOperand(1).getNode();
9228 return N0->hasOneUse() && N1->hasOneUse() &&
9229 isSignExtended(N0, DAG) && isSignExtended(N1, DAG);
9230 }
9231 return false;
9232}
9233
9234static bool isAddSubZExt(SDNode *N, SelectionDAG &DAG) {
9235 unsigned Opcode = N->getOpcode();
9236 if (Opcode == ISD::ADD || Opcode == ISD::SUB) {
9237 SDNode *N0 = N->getOperand(0).getNode();
9238 SDNode *N1 = N->getOperand(1).getNode();
9239 return N0->hasOneUse() && N1->hasOneUse() &&
9240 isZeroExtended(N0, DAG) && isZeroExtended(N1, DAG);
9241 }
9242 return false;
9243}
9244
9246 // Multiplications are only custom-lowered for 128-bit vectors so that
9247 // VMULL can be detected. Otherwise v2i64 multiplications are not legal.
9248 EVT VT = Op.getValueType();
9249 assert(VT.is128BitVector() && VT.isInteger() &&
9250 "unexpected type for custom-lowering ISD::MUL");
9251 SDNode *N0 = Op.getOperand(0).getNode();
9252 SDNode *N1 = Op.getOperand(1).getNode();
9253 unsigned NewOpc = 0;
9254 bool isMLA = false;
9255 bool isN0SExt = isSignExtended(N0, DAG);
9256 bool isN1SExt = isSignExtended(N1, DAG);
9257 if (isN0SExt && isN1SExt)
9258 NewOpc = ARMISD::VMULLs;
9259 else {
9260 bool isN0ZExt = isZeroExtended(N0, DAG);
9261 bool isN1ZExt = isZeroExtended(N1, DAG);
9262 if (isN0ZExt && isN1ZExt)
9263 NewOpc = ARMISD::VMULLu;
9264 else if (isN1SExt || isN1ZExt) {
9265 // Look for (s/zext A + s/zext B) * (s/zext C). We want to turn these
9266 // into (s/zext A * s/zext C) + (s/zext B * s/zext C)
9267 if (isN1SExt && isAddSubSExt(N0, DAG)) {
9268 NewOpc = ARMISD::VMULLs;
9269 isMLA = true;
9270 } else if (isN1ZExt && isAddSubZExt(N0, DAG)) {
9271 NewOpc = ARMISD::VMULLu;
9272 isMLA = true;
9273 } else if (isN0ZExt && isAddSubZExt(N1, DAG)) {
9274 std::swap(N0, N1);
9275 NewOpc = ARMISD::VMULLu;
9276 isMLA = true;
9277 }
9278 }
9279
9280 if (!NewOpc) {
9281 if (VT == MVT::v2i64)
9282 // Fall through to expand this. It is not legal.
9283 return SDValue();
9284 else
9285 // Other vector multiplications are legal.
9286 return Op;
9287 }
9288 }
9289
9290 // Legalize to a VMULL instruction.
9291 SDLoc DL(Op);
9292 SDValue Op0;
9293 SDValue Op1 = SkipExtensionForVMULL(N1, DAG);
9294 if (!isMLA) {
9295 Op0 = SkipExtensionForVMULL(N0, DAG);
9297 Op1.getValueType().is64BitVector() &&
9298 "unexpected types for extended operands to VMULL");
9299 return DAG.getNode(NewOpc, DL, VT, Op0, Op1);
9300 }
9301
9302 // Optimizing (zext A + zext B) * C, to (VMULL A, C) + (VMULL B, C) during
9303 // isel lowering to take advantage of no-stall back to back vmul + vmla.
9304 // vmull q0, d4, d6
9305 // vmlal q0, d5, d6
9306 // is faster than
9307 // vaddl q0, d4, d5
9308 // vmovl q1, d6
9309 // vmul q0, q0, q1
9310 SDValue N00 = SkipExtensionForVMULL(N0->getOperand(0).getNode(), DAG);
9311 SDValue N01 = SkipExtensionForVMULL(N0->getOperand(1).getNode(), DAG);
9312 EVT Op1VT = Op1.getValueType();
9313 return DAG.getNode(N0->getOpcode(), DL, VT,
9314 DAG.getNode(NewOpc, DL, VT,
9315 DAG.getNode(ISD::BITCAST, DL, Op1VT, N00), Op1),
9316 DAG.getNode(NewOpc, DL, VT,
9317 DAG.getNode(ISD::BITCAST, DL, Op1VT, N01), Op1));
9318}
9319
9321 SelectionDAG &DAG) {
9322 // TODO: Should this propagate fast-math-flags?
9323
9324 // Convert to float
9325 // float4 xf = vcvt_f32_s32(vmovl_s16(a.lo));
9326 // float4 yf = vcvt_f32_s32(vmovl_s16(b.lo));
9327 X = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, X);
9328 Y = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, Y);
9329 X = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, X);
9330 Y = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, Y);
9331 // Get reciprocal estimate.
9332 // float4 recip = vrecpeq_f32(yf);
9333 Y = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9334 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32),
9335 Y);
9336 // Because char has a smaller range than uchar, we can actually get away
9337 // without any newton steps. This requires that we use a weird bias
9338 // of 0xb000, however (again, this has been exhaustively tested).
9339 // float4 result = as_float4(as_int4(xf*recip) + 0xb000);
9340 X = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, X, Y);
9341 X = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, X);
9342 Y = DAG.getConstant(0xb000, dl, MVT::v4i32);
9343 X = DAG.getNode(ISD::ADD, dl, MVT::v4i32, X, Y);
9344 X = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, X);
9345 // Convert back to short.
9346 X = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, X);
9347 X = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, X);
9348 return X;
9349}
9350
9352 SelectionDAG &DAG) {
9353 // TODO: Should this propagate fast-math-flags?
9354
9355 SDValue N2;
9356 // Convert to float.
9357 // float4 yf = vcvt_f32_s32(vmovl_s16(y));
9358 // float4 xf = vcvt_f32_s32(vmovl_s16(x));
9359 N0 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, N0);
9360 N1 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v4i32, N1);
9361 N0 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N0);
9362 N1 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N1);
9363
9364 // Use reciprocal estimate and one refinement step.
9365 // float4 recip = vrecpeq_f32(yf);
9366 // recip *= vrecpsq_f32(yf, recip);
9367 N2 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9368 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32),
9369 N1);
9370 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9371 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32),
9372 N1, N2);
9373 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2);
9374 // Because short has a smaller range than ushort, we can actually get away
9375 // with only a single newton step. This requires that we use a weird bias
9376 // of 89, however (again, this has been exhaustively tested).
9377 // float4 result = as_float4(as_int4(xf*recip) + 0x89);
9378 N0 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N0, N2);
9379 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, N0);
9380 N1 = DAG.getConstant(0x89, dl, MVT::v4i32);
9381 N0 = DAG.getNode(ISD::ADD, dl, MVT::v4i32, N0, N1);
9382 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, N0);
9383 // Convert back to integer and return.
9384 // return vmovn_s32(vcvt_s32_f32(result));
9385 N0 = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, N0);
9386 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, N0);
9387 return N0;
9388}
9389
9391 const ARMSubtarget *ST) {
9392 EVT VT = Op.getValueType();
9393 assert((VT == MVT::v4i16 || VT == MVT::v8i8) &&
9394 "unexpected type for custom-lowering ISD::SDIV");
9395
9396 SDLoc dl(Op);
9397 SDValue N0 = Op.getOperand(0);
9398 SDValue N1 = Op.getOperand(1);
9399 SDValue N2, N3;
9400
9401 if (VT == MVT::v8i8) {
9402 N0 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v8i16, N0);
9403 N1 = DAG.getNode(ISD::SIGN_EXTEND, dl, MVT::v8i16, N1);
9404
9405 N2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0,
9406 DAG.getIntPtrConstant(4, dl));
9407 N3 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1,
9408 DAG.getIntPtrConstant(4, dl));
9409 N0 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0,
9410 DAG.getIntPtrConstant(0, dl));
9411 N1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1,
9412 DAG.getIntPtrConstant(0, dl));
9413
9414 N0 = LowerSDIV_v4i8(N0, N1, dl, DAG); // v4i16
9415 N2 = LowerSDIV_v4i8(N2, N3, dl, DAG); // v4i16
9416
9417 N0 = DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v8i16, N0, N2);
9418 N0 = LowerCONCAT_VECTORS(N0, DAG, ST);
9419
9420 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v8i8, N0);
9421 return N0;
9422 }
9423 return LowerSDIV_v4i16(N0, N1, dl, DAG);
9424}
9425
9427 const ARMSubtarget *ST) {
9428 // TODO: Should this propagate fast-math-flags?
9429 EVT VT = Op.getValueType();
9430 assert((VT == MVT::v4i16 || VT == MVT::v8i8) &&
9431 "unexpected type for custom-lowering ISD::UDIV");
9432
9433 SDLoc dl(Op);
9434 SDValue N0 = Op.getOperand(0);
9435 SDValue N1 = Op.getOperand(1);
9436 SDValue N2, N3;
9437
9438 if (VT == MVT::v8i8) {
9439 N0 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v8i16, N0);
9440 N1 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v8i16, N1);
9441
9442 N2 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0,
9443 DAG.getIntPtrConstant(4, dl));
9444 N3 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1,
9445 DAG.getIntPtrConstant(4, dl));
9446 N0 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N0,
9447 DAG.getIntPtrConstant(0, dl));
9448 N1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MVT::v4i16, N1,
9449 DAG.getIntPtrConstant(0, dl));
9450
9451 N0 = LowerSDIV_v4i16(N0, N1, dl, DAG); // v4i16
9452 N2 = LowerSDIV_v4i16(N2, N3, dl, DAG); // v4i16
9453
9454 N0 = DAG.getNode(ISD::CONCAT_VECTORS, dl, MVT::v8i16, N0, N2);
9455 N0 = LowerCONCAT_VECTORS(N0, DAG, ST);
9456
9457 N0 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v8i8,
9458 DAG.getConstant(Intrinsic::arm_neon_vqmovnsu, dl,
9459 MVT::i32),
9460 N0);
9461 return N0;
9462 }
9463
9464 // v4i16 sdiv ... Convert to float.
9465 // float4 yf = vcvt_f32_s32(vmovl_u16(y));
9466 // float4 xf = vcvt_f32_s32(vmovl_u16(x));
9467 N0 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v4i32, N0);
9468 N1 = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::v4i32, N1);
9469 N0 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N0);
9470 SDValue BN1 = DAG.getNode(ISD::SINT_TO_FP, dl, MVT::v4f32, N1);
9471
9472 // Use reciprocal estimate and two refinement steps.
9473 // float4 recip = vrecpeq_f32(yf);
9474 // recip *= vrecpsq_f32(yf, recip);
9475 // recip *= vrecpsq_f32(yf, recip);
9476 N2 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9477 DAG.getConstant(Intrinsic::arm_neon_vrecpe, dl, MVT::i32),
9478 BN1);
9479 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9480 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32),
9481 BN1, N2);
9482 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2);
9483 N1 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::v4f32,
9484 DAG.getConstant(Intrinsic::arm_neon_vrecps, dl, MVT::i32),
9485 BN1, N2);
9486 N2 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N1, N2);
9487 // Simply multiplying by the reciprocal estimate can leave us a few ulps
9488 // too low, so we add 2 ulps (exhaustive testing shows that this is enough,
9489 // and that it will never cause us to return an answer too large).
9490 // float4 result = as_float4(as_int4(xf*recip) + 2);
9491 N0 = DAG.getNode(ISD::FMUL, dl, MVT::v4f32, N0, N2);
9492 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4i32, N0);
9493 N1 = DAG.getConstant(2, dl, MVT::v4i32);
9494 N0 = DAG.getNode(ISD::ADD, dl, MVT::v4i32, N0, N1);
9495 N0 = DAG.getNode(ISD::BITCAST, dl, MVT::v4f32, N0);
9496 // Convert back to integer and return.
9497 // return vmovn_u32(vcvt_s32_f32(result));
9498 N0 = DAG.getNode(ISD::FP_TO_SINT, dl, MVT::v4i32, N0);
9499 N0 = DAG.getNode(ISD::TRUNCATE, dl, MVT::v4i16, N0);
9500 return N0;
9501}
9502
9504 unsigned Opcode, bool IsSigned) {
9505 EVT VT0 = Op.getValue(0).getValueType();
9506 EVT VT1 = Op.getValue(1).getValueType();
9507
9508 bool InvertCarry = Opcode == ARMISD::SUBE;
9509 SDValue OpLHS = Op.getOperand(0);
9510 SDValue OpRHS = Op.getOperand(1);
9511 SDValue OpCarryIn = valueToCarryFlag(Op.getOperand(2), DAG, InvertCarry);
9512
9513 SDLoc DL(Op);
9514
9515 SDValue Result = DAG.getNode(Opcode, DL, DAG.getVTList(VT0, MVT::i32), OpLHS,
9516 OpRHS, OpCarryIn);
9517
9518 SDValue OutFlag =
9519 IsSigned ? overflowFlagToValue(Result.getValue(1), VT1, DAG)
9520 : carryFlagToValue(Result.getValue(1), VT1, DAG, InvertCarry);
9521
9522 return DAG.getMergeValues({Result, OutFlag}, DL);
9523}
9524
9525SDValue ARMTargetLowering::LowerWindowsDIVLibCall(SDValue Op, SelectionDAG &DAG,
9526 bool Signed,
9527 SDValue &Chain) const {
9528 EVT VT = Op.getValueType();
9529 assert((VT == MVT::i32 || VT == MVT::i64) &&
9530 "unexpected type for custom lowering DIV");
9531 SDLoc dl(Op);
9532
9533 const auto &DL = DAG.getDataLayout();
9534 RTLIB::Libcall LC;
9535 if (Signed)
9536 LC = VT == MVT::i32 ? RTLIB::SDIVREM_I32 : RTLIB::SDIVREM_I64;
9537 else
9538 LC = VT == MVT::i32 ? RTLIB::UDIVREM_I32 : RTLIB::UDIVREM_I64;
9539
9540 RTLIB::LibcallImpl LCImpl = DAG.getLibcalls().getLibcallImpl(LC);
9541 SDValue ES = DAG.getExternalSymbol(LCImpl, getPointerTy(DL));
9542
9544
9545 for (auto AI : {1, 0}) {
9546 SDValue Operand = Op.getOperand(AI);
9547 Args.emplace_back(Operand,
9548 Operand.getValueType().getTypeForEVT(*DAG.getContext()));
9549 }
9550
9551 CallLoweringInfo CLI(DAG);
9552 CLI.setDebugLoc(dl).setChain(Chain).setCallee(
9554 VT.getTypeForEVT(*DAG.getContext()), ES, std::move(Args));
9555
9556 return LowerCallTo(CLI).first;
9557}
9558
9559// This is a code size optimisation: return the original SDIV node to
9560// DAGCombiner when we don't want to expand SDIV into a sequence of
9561// instructions, and an empty node otherwise which will cause the
9562// SDIV to be expanded in DAGCombine.
9563SDValue
9564ARMTargetLowering::BuildSDIVPow2(SDNode *N, const APInt &Divisor,
9565 SelectionDAG &DAG,
9566 SmallVectorImpl<SDNode *> &Created) const {
9567 // TODO: Support SREM
9568 if (N->getOpcode() != ISD::SDIV)
9569 return SDValue();
9570
9571 const auto &ST = DAG.getSubtarget<ARMSubtarget>();
9572 const bool MinSize = ST.hasMinSize();
9573 const bool HasDivide = ST.isThumb() ? ST.hasDivideInThumbMode()
9574 : ST.hasDivideInARMMode();
9575
9576 // Don't touch vector types; rewriting this may lead to scalarizing
9577 // the int divs.
9578 if (N->getOperand(0).getValueType().isVector())
9579 return SDValue();
9580
9581 // Bail if MinSize is not set, and also for both ARM and Thumb mode we need
9582 // hwdiv support for this to be really profitable.
9583 if (!(MinSize && HasDivide))
9584 return SDValue();
9585
9586 // ARM mode is a bit simpler than Thumb: we can handle large power
9587 // of 2 immediates with 1 mov instruction; no further checks required,
9588 // just return the sdiv node.
9589 if (!ST.isThumb())
9590 return SDValue(N, 0);
9591
9592 // In Thumb mode, immediates larger than 128 need a wide 4-byte MOV,
9593 // and thus lose the code size benefits of a MOVS that requires only 2.
9594 // TargetTransformInfo and 'getIntImmCodeSizeCost' could be helpful here,
9595 // but as it's doing exactly this, it's not worth the trouble to get TTI.
9596 if (Divisor.sgt(128))
9597 return SDValue();
9598
9599 return SDValue(N, 0);
9600}
9601
9602SDValue ARMTargetLowering::LowerDIV_Windows(SDValue Op, SelectionDAG &DAG,
9603 bool Signed) const {
9604 assert(Op.getValueType() == MVT::i32 &&
9605 "unexpected type for custom lowering DIV");
9606 SDLoc dl(Op);
9607
9608 SDValue DBZCHK = DAG.getNode(ARMISD::WIN__DBZCHK, dl, MVT::Other,
9609 DAG.getEntryNode(), Op.getOperand(1));
9610
9611 return LowerWindowsDIVLibCall(Op, DAG, Signed, DBZCHK);
9612}
9613
9615 SDLoc DL(N);
9616 SDValue Op = N->getOperand(1);
9617 if (N->getValueType(0) == MVT::i32)
9618 return DAG.getNode(ARMISD::WIN__DBZCHK, DL, MVT::Other, InChain, Op);
9619 SDValue Lo, Hi;
9620 std::tie(Lo, Hi) = DAG.SplitScalar(Op, DL, MVT::i32, MVT::i32);
9621 return DAG.getNode(ARMISD::WIN__DBZCHK, DL, MVT::Other, InChain,
9622 DAG.getNode(ISD::OR, DL, MVT::i32, Lo, Hi));
9623}
9624
9625void ARMTargetLowering::ExpandDIV_Windows(
9626 SDValue Op, SelectionDAG &DAG, bool Signed,
9628 const auto &DL = DAG.getDataLayout();
9629
9630 assert(Op.getValueType() == MVT::i64 &&
9631 "unexpected type for custom lowering DIV");
9632 SDLoc dl(Op);
9633
9634 SDValue DBZCHK = WinDBZCheckDenominator(DAG, Op.getNode(), DAG.getEntryNode());
9635
9636 SDValue Result = LowerWindowsDIVLibCall(Op, DAG, Signed, DBZCHK);
9637
9638 SDValue Lower = DAG.getNode(ISD::TRUNCATE, dl, MVT::i32, Result);
9639 SDValue Upper = DAG.getNode(ISD::SRL, dl, MVT::i64, Result,
9640 DAG.getConstant(32, dl, getPointerTy(DL)));
9641 Upper = DAG.getNode(ISD::TRUNCATE, dl, MVT::i32, Upper);
9642
9643 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lower, Upper));
9644}
9645
9646std::pair<SDValue, SDValue>
9647ARMTargetLowering::LowerAEABIUnalignedLoad(SDValue Op,
9648 SelectionDAG &DAG) const {
9649 // If we have an unaligned load from a i32 or i64 that would normally be
9650 // split into separate ldrb's, we can use the __aeabi_uread4/__aeabi_uread8
9651 // functions instead.
9652 LoadSDNode *LD = cast<LoadSDNode>(Op.getNode());
9653 EVT MemVT = LD->getMemoryVT();
9654 if (MemVT != MVT::i32 && MemVT != MVT::i64)
9655 return std::make_pair(SDValue(), SDValue());
9656
9657 const auto &MF = DAG.getMachineFunction();
9658 unsigned AS = LD->getAddressSpace();
9659 Align Alignment = LD->getAlign();
9660 const DataLayout &DL = DAG.getDataLayout();
9661 bool AllowsUnaligned = Subtarget->allowsUnalignedMem();
9662 RTLIB::Libcall LC =
9663 (MemVT == MVT::i32) ? RTLIB::AEABI_UREAD4 : RTLIB::AEABI_UREAD8;
9664
9665 if (MF.getFunction().hasMinSize() && !AllowsUnaligned &&
9666 Alignment <= llvm::Align(2) && DAG.getLibcalls().getLibcallImpl(LC)) {
9667 MakeLibCallOptions Opts;
9668 SDLoc dl(Op);
9669
9670 auto Pair = makeLibCall(DAG, LC, MemVT.getSimpleVT(), LD->getBasePtr(),
9671 Opts, dl, LD->getChain());
9672
9673 // If necessary, extend the node to 64bit
9674 if (LD->getExtensionType() != ISD::NON_EXTLOAD) {
9675 unsigned ExtType = LD->getExtensionType() == ISD::SEXTLOAD
9678 SDValue EN = DAG.getNode(ExtType, dl, LD->getValueType(0), Pair.first);
9679 Pair.first = EN;
9680 }
9681 return Pair;
9682 }
9683
9684 // Default expand to individual loads
9685 if (!allowsMemoryAccess(*DAG.getContext(), DL, MemVT, AS, Alignment))
9686 return expandUnalignedLoad(LD, DAG);
9687 return std::make_pair(SDValue(), SDValue());
9688}
9689
9690SDValue ARMTargetLowering::LowerAEABIUnalignedStore(SDValue Op,
9691 SelectionDAG &DAG) const {
9692 // If we have an unaligned store to a i32 or i64 that would normally be
9693 // split into separate ldrb's, we can use the __aeabi_uwrite4/__aeabi_uwrite8
9694 // functions instead.
9695 StoreSDNode *ST = cast<StoreSDNode>(Op.getNode());
9696 EVT MemVT = ST->getMemoryVT();
9697 if (MemVT != MVT::i32 && MemVT != MVT::i64)
9698 return SDValue();
9699
9700 const auto &MF = DAG.getMachineFunction();
9701 unsigned AS = ST->getAddressSpace();
9702 Align Alignment = ST->getAlign();
9703 const DataLayout &DL = DAG.getDataLayout();
9704 bool AllowsUnaligned = Subtarget->allowsUnalignedMem();
9705 RTLIB::Libcall LC =
9706 (MemVT == MVT::i32) ? RTLIB::AEABI_UWRITE4 : RTLIB::AEABI_UWRITE8;
9707
9708 if (MF.getFunction().hasMinSize() && !AllowsUnaligned &&
9709 Alignment <= llvm::Align(2) && DAG.getLibcalls().getLibcallImpl(LC)) {
9710
9711 SDLoc dl(Op);
9712
9713 // If necessary, trunc the value to 32bit
9714 SDValue StoreVal = ST->getOperand(1);
9715 if (ST->isTruncatingStore())
9716 StoreVal = DAG.getNode(ISD::TRUNCATE, dl, MemVT, ST->getOperand(1));
9717
9718 MakeLibCallOptions Opts;
9719 auto CallResult =
9720 makeLibCall(DAG, LC, MVT::isVoid, {StoreVal, ST->getBasePtr()}, Opts,
9721 dl, ST->getChain());
9722
9723 return CallResult.second;
9724 }
9725
9726 // Default expand to individual stores
9727 if (!allowsMemoryAccess(*DAG.getContext(), DL, MemVT, AS, Alignment))
9728 return expandUnalignedStore(ST, DAG);
9729 return SDValue();
9730}
9731
9733 LoadSDNode *LD = cast<LoadSDNode>(Op.getNode());
9734 EVT MemVT = LD->getMemoryVT();
9735 assert((MemVT == MVT::v2i1 || MemVT == MVT::v4i1 || MemVT == MVT::v8i1 ||
9736 MemVT == MVT::v16i1) &&
9737 "Expected a predicate type!");
9738 assert(MemVT == Op.getValueType());
9739 assert(LD->getExtensionType() == ISD::NON_EXTLOAD &&
9740 "Expected a non-extending load");
9741 assert(LD->isUnindexed() && "Expected a unindexed load");
9742
9743 // The basic MVE VLDR on a v2i1/v4i1/v8i1 actually loads the entire 16bit
9744 // predicate, with the "v4i1" bits spread out over the 16 bits loaded. We
9745 // need to make sure that 8/4/2 bits are actually loaded into the correct
9746 // place, which means loading the value and then shuffling the values into
9747 // the bottom bits of the predicate.
9748 // Equally, VLDR for an v16i1 will actually load 32bits (so will be incorrect
9749 // for BE).
9750 // Speaking of BE, apparently the rest of llvm will assume a reverse order to
9751 // a natural VMSR(load), so needs to be reversed.
9752
9753 SDLoc dl(Op);
9754 SDValue Load = DAG.getExtLoad(
9755 ISD::EXTLOAD, dl, MVT::i32, LD->getChain(), LD->getBasePtr(),
9757 LD->getMemOperand());
9758 SDValue Val = Load;
9759 if (DAG.getDataLayout().isBigEndian())
9760 Val = DAG.getNode(ISD::SRL, dl, MVT::i32,
9761 DAG.getNode(ISD::BITREVERSE, dl, MVT::i32, Load),
9762 DAG.getConstant(32 - MemVT.getSizeInBits(), dl, MVT::i32));
9763 SDValue Pred = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::v16i1, Val);
9764 if (MemVT != MVT::v16i1)
9765 Pred = DAG.getNode(ISD::EXTRACT_SUBVECTOR, dl, MemVT, Pred,
9766 DAG.getConstant(0, dl, MVT::i32));
9767 return DAG.getMergeValues({Pred, Load.getValue(1)}, dl);
9768}
9769
9770void ARMTargetLowering::LowerLOAD(SDNode *N, SmallVectorImpl<SDValue> &Results,
9771 SelectionDAG &DAG) const {
9772 LoadSDNode *LD = cast<LoadSDNode>(N);
9773 EVT MemVT = LD->getMemoryVT();
9774
9775 if (MemVT == MVT::i64 && Subtarget->hasV5TEOps() &&
9776 !Subtarget->isThumb1Only() && LD->isVolatile() &&
9777 LD->getAlign() >= Subtarget->getDualLoadStoreAlignment()) {
9778 assert(LD->isUnindexed() && "Loads should be unindexed at this point.");
9779 SDLoc dl(N);
9780 SDValue Result = DAG.getMemIntrinsicNode(
9781 ARMISD::LDRD, dl, DAG.getVTList({MVT::i32, MVT::i32, MVT::Other}),
9782 {LD->getChain(), LD->getBasePtr()}, MemVT, LD->getMemOperand());
9783 SDValue Lo = Result.getValue(DAG.getDataLayout().isLittleEndian() ? 0 : 1);
9784 SDValue Hi = Result.getValue(DAG.getDataLayout().isLittleEndian() ? 1 : 0);
9785 SDValue Pair = DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Lo, Hi);
9786 Results.append({Pair, Result.getValue(2)});
9787 } else if (MemVT == MVT::i32 || MemVT == MVT::i64) {
9788 auto Pair = LowerAEABIUnalignedLoad(SDValue(N, 0), DAG);
9789 if (Pair.first) {
9790 Results.push_back(Pair.first);
9791 Results.push_back(Pair.second);
9792 }
9793 }
9794}
9795
9797 StoreSDNode *ST = cast<StoreSDNode>(Op.getNode());
9798 EVT MemVT = ST->getMemoryVT();
9799 assert((MemVT == MVT::v2i1 || MemVT == MVT::v4i1 || MemVT == MVT::v8i1 ||
9800 MemVT == MVT::v16i1) &&
9801 "Expected a predicate type!");
9802 assert(MemVT == ST->getValue().getValueType());
9803 assert(!ST->isTruncatingStore() && "Expected a non-extending store");
9804 assert(ST->isUnindexed() && "Expected a unindexed store");
9805
9806 // Only store the v2i1 or v4i1 or v8i1 worth of bits, via a buildvector with
9807 // top bits unset and a scalar store.
9808 SDLoc dl(Op);
9809 SDValue Build = ST->getValue();
9810 if (MemVT != MVT::v16i1) {
9812 for (unsigned I = 0; I < MemVT.getVectorNumElements(); I++) {
9813 unsigned Elt = DAG.getDataLayout().isBigEndian()
9814 ? MemVT.getVectorNumElements() - I - 1
9815 : I;
9816 Ops.push_back(DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::i32, Build,
9817 DAG.getConstant(Elt, dl, MVT::i32)));
9818 }
9819 for (unsigned I = MemVT.getVectorNumElements(); I < 16; I++)
9820 Ops.push_back(DAG.getUNDEF(MVT::i32));
9821 Build = DAG.getNode(ISD::BUILD_VECTOR, dl, MVT::v16i1, Ops);
9822 }
9823 SDValue GRP = DAG.getNode(ARMISD::PREDICATE_CAST, dl, MVT::i32, Build);
9824 if (MemVT == MVT::v16i1 && DAG.getDataLayout().isBigEndian())
9825 GRP = DAG.getNode(ISD::SRL, dl, MVT::i32,
9826 DAG.getNode(ISD::BITREVERSE, dl, MVT::i32, GRP),
9827 DAG.getConstant(16, dl, MVT::i32));
9828 return DAG.getTruncStore(
9829 ST->getChain(), dl, GRP, ST->getBasePtr(),
9831 ST->getMemOperand());
9832}
9833
9834SDValue ARMTargetLowering::LowerSTORE(SDValue Op, SelectionDAG &DAG,
9835 const ARMSubtarget *Subtarget) const {
9836 StoreSDNode *ST = cast<StoreSDNode>(Op.getNode());
9837 EVT MemVT = ST->getMemoryVT();
9838
9839 if (MemVT == MVT::i64 && Subtarget->hasV5TEOps() &&
9840 !Subtarget->isThumb1Only() && ST->isVolatile() &&
9841 ST->getAlign() >= Subtarget->getDualLoadStoreAlignment()) {
9842 assert(ST->isUnindexed() && "Stores should be unindexed at this point.");
9843 SDNode *N = Op.getNode();
9844 SDLoc dl(N);
9845
9846 SDValue Lo = DAG.getNode(
9847 ISD::EXTRACT_ELEMENT, dl, MVT::i32, ST->getValue(),
9848 DAG.getTargetConstant(DAG.getDataLayout().isLittleEndian() ? 0 : 1, dl,
9849 MVT::i32));
9850 SDValue Hi = DAG.getNode(
9851 ISD::EXTRACT_ELEMENT, dl, MVT::i32, ST->getValue(),
9852 DAG.getTargetConstant(DAG.getDataLayout().isLittleEndian() ? 1 : 0, dl,
9853 MVT::i32));
9854
9855 return DAG.getMemIntrinsicNode(ARMISD::STRD, dl, DAG.getVTList(MVT::Other),
9856 {ST->getChain(), Lo, Hi, ST->getBasePtr()},
9857 MemVT, ST->getMemOperand());
9858 } else if (Subtarget->hasMVEIntegerOps() &&
9859 ((MemVT == MVT::v2i1 || MemVT == MVT::v4i1 || MemVT == MVT::v8i1 ||
9860 MemVT == MVT::v16i1))) {
9861 return LowerPredicateStore(Op, DAG);
9862 } else if (MemVT == MVT::i32 || MemVT == MVT::i64) {
9863 return LowerAEABIUnalignedStore(Op, DAG);
9864 }
9865 return SDValue();
9866}
9867
9868static bool isZeroVector(SDValue N) {
9869 return (ISD::isBuildVectorAllZeros(N.getNode()) ||
9870 (N->getOpcode() == ARMISD::VMOVIMM &&
9871 isNullConstant(N->getOperand(0))));
9872}
9873
9876 MVT VT = Op.getSimpleValueType();
9877 SDValue Mask = N->getMask();
9878 SDValue PassThru = N->getPassThru();
9879 SDLoc dl(Op);
9880
9881 if (isZeroVector(PassThru))
9882 return Op;
9883
9884 // MVE Masked loads use zero as the passthru value. Here we convert undef to
9885 // zero too, and other values are lowered to a select.
9886 SDValue ZeroVec = DAG.getNode(ARMISD::VMOVIMM, dl, VT,
9887 DAG.getTargetConstant(0, dl, MVT::i32));
9888 SDValue NewLoad = DAG.getMaskedLoad(
9889 VT, dl, N->getChain(), N->getBasePtr(), N->getOffset(), Mask, ZeroVec,
9890 N->getMemoryVT(), N->getMemOperand(), N->getAddressingMode(),
9891 N->getExtensionType(), N->isExpandingLoad());
9892 SDValue Combo = NewLoad;
9893 bool PassThruIsCastZero = (PassThru.getOpcode() == ISD::BITCAST ||
9894 PassThru.getOpcode() == ARMISD::VECTOR_REG_CAST) &&
9895 isZeroVector(PassThru->getOperand(0));
9896 if (!PassThru.isUndef() && !PassThruIsCastZero)
9897 Combo = DAG.getNode(ISD::VSELECT, dl, VT, Mask, NewLoad, PassThru);
9898 return DAG.getMergeValues({Combo, NewLoad.getValue(1)}, dl);
9899}
9900
9902 const ARMSubtarget *ST) {
9903 if (!ST->hasMVEIntegerOps())
9904 return SDValue();
9905
9906 SDLoc dl(Op);
9907 unsigned BaseOpcode = 0;
9908 switch (Op->getOpcode()) {
9909 default: llvm_unreachable("Expected VECREDUCE opcode");
9910 case ISD::VECREDUCE_FADD: BaseOpcode = ISD::FADD; break;
9911 case ISD::VECREDUCE_FMUL: BaseOpcode = ISD::FMUL; break;
9912 case ISD::VECREDUCE_MUL: BaseOpcode = ISD::MUL; break;
9913 case ISD::VECREDUCE_AND: BaseOpcode = ISD::AND; break;
9914 case ISD::VECREDUCE_OR: BaseOpcode = ISD::OR; break;
9915 case ISD::VECREDUCE_XOR: BaseOpcode = ISD::XOR; break;
9916 case ISD::VECREDUCE_FMAX: BaseOpcode = ISD::FMAXNUM; break;
9917 case ISD::VECREDUCE_FMIN: BaseOpcode = ISD::FMINNUM; break;
9918 }
9919
9920 SDValue Op0 = Op->getOperand(0);
9921 EVT VT = Op0.getValueType();
9922 EVT EltVT = VT.getVectorElementType();
9923 unsigned NumElts = VT.getVectorNumElements();
9924 unsigned NumActiveLanes = NumElts;
9925
9926 assert((NumActiveLanes == 16 || NumActiveLanes == 8 || NumActiveLanes == 4 ||
9927 NumActiveLanes == 2) &&
9928 "Only expected a power 2 vector size");
9929
9930 // Use Mul(X, Rev(X)) until 4 items remain. Going down to 4 vector elements
9931 // allows us to easily extract vector elements from the lanes.
9932 while (NumActiveLanes > 4) {
9933 unsigned RevOpcode = NumActiveLanes == 16 ? ARMISD::VREV16 : ARMISD::VREV32;
9934 SDValue Rev = DAG.getNode(RevOpcode, dl, VT, Op0);
9935 Op0 = DAG.getNode(BaseOpcode, dl, VT, Op0, Rev);
9936 NumActiveLanes /= 2;
9937 }
9938
9939 SDValue Res;
9940 if (NumActiveLanes == 4) {
9941 // The remaining 4 elements are summed sequentially
9942 SDValue Ext0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9943 DAG.getConstant(0 * NumElts / 4, dl, MVT::i32));
9944 SDValue Ext1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9945 DAG.getConstant(1 * NumElts / 4, dl, MVT::i32));
9946 SDValue Ext2 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9947 DAG.getConstant(2 * NumElts / 4, dl, MVT::i32));
9948 SDValue Ext3 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9949 DAG.getConstant(3 * NumElts / 4, dl, MVT::i32));
9950 SDValue Res0 = DAG.getNode(BaseOpcode, dl, EltVT, Ext0, Ext1, Op->getFlags());
9951 SDValue Res1 = DAG.getNode(BaseOpcode, dl, EltVT, Ext2, Ext3, Op->getFlags());
9952 Res = DAG.getNode(BaseOpcode, dl, EltVT, Res0, Res1, Op->getFlags());
9953 } else {
9954 SDValue Ext0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9955 DAG.getConstant(0, dl, MVT::i32));
9956 SDValue Ext1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
9957 DAG.getConstant(1, dl, MVT::i32));
9958 Res = DAG.getNode(BaseOpcode, dl, EltVT, Ext0, Ext1, Op->getFlags());
9959 }
9960
9961 // Result type may be wider than element type.
9962 if (EltVT != Op->getValueType(0))
9963 Res = DAG.getNode(ISD::ANY_EXTEND, dl, Op->getValueType(0), Res);
9964 return Res;
9965}
9966
9968 const ARMSubtarget *ST) {
9969 if (!ST->hasMVEFloatOps())
9970 return SDValue();
9971 return LowerVecReduce(Op, DAG, ST);
9972}
9973
9975 const ARMSubtarget *ST) {
9976 if (!ST->hasNEON())
9977 return SDValue();
9978
9979 SDLoc dl(Op);
9980 SDValue Op0 = Op->getOperand(0);
9981 EVT VT = Op0.getValueType();
9982 EVT EltVT = VT.getVectorElementType();
9983
9984 unsigned PairwiseIntrinsic = 0;
9985 switch (Op->getOpcode()) {
9986 default:
9987 llvm_unreachable("Expected VECREDUCE opcode");
9989 PairwiseIntrinsic = Intrinsic::arm_neon_vpminu;
9990 break;
9992 PairwiseIntrinsic = Intrinsic::arm_neon_vpmaxu;
9993 break;
9995 PairwiseIntrinsic = Intrinsic::arm_neon_vpmins;
9996 break;
9998 PairwiseIntrinsic = Intrinsic::arm_neon_vpmaxs;
9999 break;
10000 }
10001 SDValue PairwiseOp = DAG.getConstant(PairwiseIntrinsic, dl, MVT::i32);
10002
10003 unsigned NumElts = VT.getVectorNumElements();
10004 unsigned NumActiveLanes = NumElts;
10005
10006 assert((NumActiveLanes == 16 || NumActiveLanes == 8 || NumActiveLanes == 4 ||
10007 NumActiveLanes == 2) &&
10008 "Only expected a power 2 vector size");
10009
10010 // Split 128-bit vectors, since vpmin/max takes 2 64-bit vectors.
10011 if (VT.is128BitVector()) {
10012 SDValue Lo, Hi;
10013 std::tie(Lo, Hi) = DAG.SplitVector(Op0, dl);
10014 VT = Lo.getValueType();
10015 Op0 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, {PairwiseOp, Lo, Hi});
10016 NumActiveLanes /= 2;
10017 }
10018
10019 // Use pairwise reductions until one lane remains
10020 while (NumActiveLanes > 1) {
10021 Op0 = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, {PairwiseOp, Op0, Op0});
10022 NumActiveLanes /= 2;
10023 }
10024
10025 SDValue Res = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, EltVT, Op0,
10026 DAG.getConstant(0, dl, MVT::i32));
10027
10028 // Result type may be wider than element type.
10029 if (EltVT != Op.getValueType()) {
10030 unsigned Extend = 0;
10031 switch (Op->getOpcode()) {
10032 default:
10033 llvm_unreachable("Expected VECREDUCE opcode");
10036 Extend = ISD::ZERO_EXTEND;
10037 break;
10040 Extend = ISD::SIGN_EXTEND;
10041 break;
10042 }
10043 Res = DAG.getNode(Extend, dl, Op.getValueType(), Res);
10044 }
10045 return Res;
10046}
10047
10049 if (isStrongerThanMonotonic(cast<AtomicSDNode>(Op)->getSuccessOrdering()))
10050 // Acquire/Release load/store is not legal for targets without a dmb or
10051 // equivalent available.
10052 return SDValue();
10053
10054 // Monotonic load/store is legal for all targets.
10055 return Op;
10056}
10057
10060 SelectionDAG &DAG,
10061 const ARMSubtarget *Subtarget) {
10062 SDLoc DL(N);
10063 // Under Power Management extensions, the cycle-count is:
10064 // mrc p15, #0, <Rt>, c9, c13, #0
10065 SDValue Ops[] = { N->getOperand(0), // Chain
10066 DAG.getTargetConstant(Intrinsic::arm_mrc, DL, MVT::i32),
10067 DAG.getTargetConstant(15, DL, MVT::i32),
10068 DAG.getTargetConstant(0, DL, MVT::i32),
10069 DAG.getTargetConstant(9, DL, MVT::i32),
10070 DAG.getTargetConstant(13, DL, MVT::i32),
10071 DAG.getTargetConstant(0, DL, MVT::i32)
10072 };
10073
10074 SDValue Cycles32 = DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL,
10075 DAG.getVTList(MVT::i32, MVT::Other), Ops);
10076 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, Cycles32,
10077 DAG.getConstant(0, DL, MVT::i32)));
10078 Results.push_back(Cycles32.getValue(1));
10079}
10080
10082 SDValue V1) {
10083 SDLoc dl(V0.getNode());
10084 SDValue RegClass =
10085 DAG.getTargetConstant(ARM::GPRPairRegClassID, dl, MVT::i32);
10086 SDValue SubReg0 = DAG.getTargetConstant(ARM::gsub_0, dl, MVT::i32);
10087 SDValue SubReg1 = DAG.getTargetConstant(ARM::gsub_1, dl, MVT::i32);
10088 const SDValue Ops[] = {RegClass, V0, SubReg0, V1, SubReg1};
10089 return SDValue(
10090 DAG.getMachineNode(TargetOpcode::REG_SEQUENCE, dl, MVT::Untyped, Ops), 0);
10091}
10092
10094 SDLoc dl(V.getNode());
10095 auto [VLo, VHi] = DAG.SplitScalar(V, dl, MVT::i32, MVT::i32);
10096 bool isBigEndian = DAG.getDataLayout().isBigEndian();
10097 if (isBigEndian)
10098 std::swap(VLo, VHi);
10099 return createGPRPairNode2xi32(DAG, VLo, VHi);
10100}
10101
10104 SelectionDAG &DAG) {
10105 assert(N->getValueType(0) == MVT::i64 &&
10106 "AtomicCmpSwap on types less than 64 should be legal");
10107 SDValue Ops[] = {
10108 createGPRPairNode2xi32(DAG, N->getOperand(1),
10109 DAG.getUNDEF(MVT::i32)), // pointer, temp
10110 createGPRPairNodei64(DAG, N->getOperand(2)), // expected
10111 createGPRPairNodei64(DAG, N->getOperand(3)), // new
10112 N->getOperand(0), // chain in
10113 };
10114 SDNode *CmpSwap = DAG.getMachineNode(
10115 ARM::CMP_SWAP_64, SDLoc(N),
10116 DAG.getVTList(MVT::Untyped, MVT::Untyped, MVT::Other), Ops);
10117
10118 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
10119 DAG.setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp});
10120
10121 bool isBigEndian = DAG.getDataLayout().isBigEndian();
10122
10123 SDValue Lo =
10124 DAG.getTargetExtractSubreg(isBigEndian ? ARM::gsub_1 : ARM::gsub_0,
10125 SDLoc(N), MVT::i32, SDValue(CmpSwap, 0));
10126 SDValue Hi =
10127 DAG.getTargetExtractSubreg(isBigEndian ? ARM::gsub_0 : ARM::gsub_1,
10128 SDLoc(N), MVT::i32, SDValue(CmpSwap, 0));
10129 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, SDLoc(N), MVT::i64, Lo, Hi));
10130 Results.push_back(SDValue(CmpSwap, 2));
10131}
10132
10133SDValue ARMTargetLowering::LowerFSETCC(SDValue Op, SelectionDAG &DAG) const {
10134 SDLoc dl(Op);
10135 EVT VT = Op.getValueType();
10136 SDValue Chain = Op.getOperand(0);
10137 SDValue LHS = Op.getOperand(1);
10138 SDValue RHS = Op.getOperand(2);
10139 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(3))->get();
10140 bool IsSignaling = Op.getOpcode() == ISD::STRICT_FSETCCS;
10141
10142 // If we don't have instructions of this float type then soften to a libcall
10143 // and use SETCC instead.
10144 if (isUnsupportedFloatingType(LHS.getValueType())) {
10145 softenSetCCOperands(DAG, LHS.getValueType(), LHS, RHS, CC, dl, LHS, RHS,
10146 Chain, IsSignaling);
10147 if (!RHS.getNode()) {
10148 RHS = DAG.getConstant(0, dl, LHS.getValueType());
10149 CC = ISD::SETNE;
10150 }
10151 SDValue Result = DAG.getNode(ISD::SETCC, dl, VT, LHS, RHS,
10152 DAG.getCondCode(CC));
10153 return DAG.getMergeValues({Result, Chain}, dl);
10154 }
10155
10156 ARMCC::CondCodes CondCode, CondCode2;
10157 FPCCToARMCC(CC, CondCode, CondCode2);
10158
10159 SDValue True = DAG.getConstant(1, dl, VT);
10160 SDValue False = DAG.getConstant(0, dl, VT);
10161 SDValue ARMcc = DAG.getConstant(CondCode, dl, MVT::i32);
10162 SDValue Cmp = getVFPCmp(LHS, RHS, DAG, dl, IsSignaling);
10163 SDValue Result = getCMOV(dl, VT, False, True, ARMcc, Cmp, DAG);
10164 if (CondCode2 != ARMCC::AL) {
10165 ARMcc = DAG.getConstant(CondCode2, dl, MVT::i32);
10166 Result = getCMOV(dl, VT, Result, True, ARMcc, Cmp, DAG);
10167 }
10168 return DAG.getMergeValues({Result, Chain}, dl);
10169}
10170
10171SDValue ARMTargetLowering::LowerSPONENTRY(SDValue Op, SelectionDAG &DAG) const {
10172 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
10173
10174 EVT VT = getPointerTy(DAG.getDataLayout());
10175 int FI = MFI.CreateFixedObject(4, 0, false);
10176 return DAG.getFrameIndex(FI, VT);
10177}
10178
10179SDValue ARMTargetLowering::LowerFP_TO_BF16(SDValue Op,
10180 SelectionDAG &DAG) const {
10181 SDLoc DL(Op);
10182 MakeLibCallOptions CallOptions;
10183 MVT SVT = Op.getOperand(0).getSimpleValueType();
10184 RTLIB::Libcall LC = RTLIB::getFPROUND(SVT, MVT::bf16);
10185 SDValue Res =
10186 makeLibCall(DAG, LC, MVT::f32, Op.getOperand(0), CallOptions, DL).first;
10187 return DAG.getBitcast(MVT::i32, Res);
10188}
10189
10190SDValue ARMTargetLowering::LowerCMP(SDValue Op, SelectionDAG &DAG) const {
10191 SDLoc dl(Op);
10192 SDValue LHS = Op.getOperand(0);
10193 SDValue RHS = Op.getOperand(1);
10194
10195 // Determine if this is signed or unsigned comparison
10196 bool IsSigned = (Op.getOpcode() == ISD::SCMP);
10197
10198 // Special case for Thumb1 UCMP only
10199 if (!IsSigned && Subtarget->isThumb1Only()) {
10200 // For Thumb unsigned comparison, use this sequence:
10201 // subs r2, r0, r1 ; r2 = LHS - RHS, sets flags
10202 // sbc r2, r2 ; r2 = r2 - r2 - !carry
10203 // cmp r1, r0 ; compare RHS with LHS
10204 // sbc r1, r1 ; r1 = r1 - r1 - !carry
10205 // subs r0, r2, r1 ; r0 = r2 - r1 (final result)
10206
10207 // First subtraction: LHS - RHS
10208 SDValue Sub1WithFlags = DAG.getNode(
10209 ARMISD::SUBC, dl, DAG.getVTList(MVT::i32, FlagsVT), LHS, RHS);
10210 SDValue Sub1Result = Sub1WithFlags.getValue(0);
10211 SDValue Flags1 = Sub1WithFlags.getValue(1);
10212
10213 // SUBE: Sub1Result - Sub1Result - !carry
10214 // This gives 0 if LHS >= RHS (unsigned), -1 if LHS < RHS (unsigned)
10215 SDValue Sbc1 =
10216 DAG.getNode(ARMISD::SUBE, dl, DAG.getVTList(MVT::i32, FlagsVT),
10217 Sub1Result, Sub1Result, Flags1);
10218 SDValue Sbc1Result = Sbc1.getValue(0);
10219
10220 // Second comparison: RHS vs LHS (reverse comparison)
10221 SDValue CmpFlags = DAG.getNode(ARMISD::CMP, dl, FlagsVT, RHS, LHS);
10222
10223 // SUBE: RHS - RHS - !carry
10224 // This gives 0 if RHS <= LHS (unsigned), -1 if RHS > LHS (unsigned)
10225 SDValue Sbc2 = DAG.getNode(
10226 ARMISD::SUBE, dl, DAG.getVTList(MVT::i32, FlagsVT), RHS, RHS, CmpFlags);
10227 SDValue Sbc2Result = Sbc2.getValue(0);
10228
10229 // Final subtraction: Sbc1Result - Sbc2Result (no flags needed)
10230 SDValue Result =
10231 DAG.getNode(ISD::SUB, dl, MVT::i32, Sbc1Result, Sbc2Result);
10232 if (Op.getValueType() != MVT::i32)
10233 Result = DAG.getSExtOrTrunc(Result, dl, Op.getValueType());
10234
10235 return Result;
10236 }
10237
10238 // For the ARM assembly pattern:
10239 // subs r0, r0, r1 ; subtract RHS from LHS and set flags
10240 // movgt r0, #1 ; if LHS > RHS, set result to 1 (GT for signed, HI for
10241 // unsigned) mvnlt r0, #0 ; if LHS < RHS, set result to -1 (LT for
10242 // signed, LO for unsigned)
10243 // ; if LHS == RHS, result remains 0 from the subs
10244
10245 // Optimization: if RHS is a subtraction against 0, use ADDC instead of SUBC
10246 unsigned Opcode = ARMISD::SUBC;
10247
10248 // Check if RHS is a subtraction against 0: (0 - X)
10249 if (RHS.getOpcode() == ISD::SUB) {
10250 SDValue SubLHS = RHS.getOperand(0);
10251 SDValue SubRHS = RHS.getOperand(1);
10252
10253 // Check if it's 0 - X
10254 if (isNullConstant(SubLHS)) {
10255 bool CanUseAdd = false;
10256 if (IsSigned) {
10257 // For SCMP: only if X is known to never be INT_MIN (to avoid overflow)
10258 if (RHS->getFlags().hasNoSignedWrap() || !DAG.computeKnownBits(SubRHS)
10260 .isMinSignedValue()) {
10261 CanUseAdd = true;
10262 }
10263 } else {
10264 // For UCMP: only if X is known to never be zero
10265 if (DAG.isKnownNeverZero(SubRHS)) {
10266 CanUseAdd = true;
10267 }
10268 }
10269
10270 if (CanUseAdd) {
10271 Opcode = ARMISD::ADDC;
10272 RHS = SubRHS; // Replace RHS with X, so we do LHS + X instead of
10273 // LHS - (0 - X)
10274 }
10275 }
10276 }
10277
10278 // Generate the operation with flags
10279 SDValue OpWithFlags =
10280 DAG.getNode(Opcode, dl, DAG.getVTList(MVT::i32, FlagsVT), LHS, RHS);
10281
10282 SDValue OpResult = OpWithFlags.getValue(0);
10283 SDValue Flags = OpWithFlags.getValue(1);
10284
10285 // Constants for conditional moves
10286 SDValue One = DAG.getConstant(1, dl, MVT::i32);
10287 SDValue MinusOne = DAG.getAllOnesConstant(dl, MVT::i32);
10288
10289 // Select condition codes based on signed vs unsigned
10290 ARMCC::CondCodes GTCond = IsSigned ? ARMCC::GT : ARMCC::HI;
10291 ARMCC::CondCodes LTCond = IsSigned ? ARMCC::LT : ARMCC::LO;
10292
10293 // First conditional move: if greater than, set to 1
10294 SDValue GTCondValue = DAG.getConstant(GTCond, dl, MVT::i32);
10295 SDValue Result1 = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, OpResult, One,
10296 GTCondValue, Flags);
10297
10298 // Second conditional move: if less than, set to -1
10299 SDValue LTCondValue = DAG.getConstant(LTCond, dl, MVT::i32);
10300 SDValue Result2 = DAG.getNode(ARMISD::CMOV, dl, MVT::i32, Result1, MinusOne,
10301 LTCondValue, Flags);
10302
10303 if (Op.getValueType() != MVT::i32)
10304 Result2 = DAG.getSExtOrTrunc(Result2, dl, Op.getValueType());
10305
10306 return Result2;
10307}
10308
10310 LLVM_DEBUG(dbgs() << "Lowering node: "; Op.dump());
10311 switch (Op.getOpcode()) {
10312 default: llvm_unreachable("Don't know how to custom lower this!");
10313 case ISD::WRITE_REGISTER: return LowerWRITE_REGISTER(Op, DAG);
10314 case ISD::ConstantPool: return LowerConstantPool(Op, DAG);
10315 case ISD::BlockAddress: return LowerBlockAddress(Op, DAG);
10316 case ISD::GlobalAddress: return LowerGlobalAddress(Op, DAG);
10317 case ISD::GlobalTLSAddress: return LowerGlobalTLSAddress(Op, DAG);
10318 case ISD::SELECT: return LowerSELECT(Op, DAG);
10319 case ISD::SELECT_CC: return LowerSELECT_CC(Op, DAG);
10320 case ISD::BRCOND: return LowerBRCOND(Op, DAG);
10321 case ISD::BR_CC: return LowerBR_CC(Op, DAG);
10322 case ISD::BR_JT: return LowerBR_JT(Op, DAG);
10323 case ISD::VASTART: return LowerVASTART(Op, DAG);
10324 case ISD::ATOMIC_FENCE: return LowerATOMIC_FENCE(Op, DAG, Subtarget);
10325 case ISD::PREFETCH: return LowerPREFETCH(Op, DAG, Subtarget);
10328 case ISD::SINT_TO_FP:
10329 case ISD::UINT_TO_FP: return LowerINT_TO_FP(Op, DAG);
10332 case ISD::FP_TO_SINT:
10333 case ISD::FP_TO_UINT: return LowerFP_TO_INT(Op, DAG);
10335 case ISD::FP_TO_UINT_SAT: return LowerFP_TO_INT_SAT(Op, DAG, Subtarget);
10336 case ISD::FCOPYSIGN: return LowerFCOPYSIGN(Op, DAG);
10337 case ISD::RETURNADDR: return LowerRETURNADDR(Op, DAG);
10338 case ISD::FRAMEADDR: return LowerFRAMEADDR(Op, DAG);
10339 case ISD::EH_SJLJ_SETJMP: return LowerEH_SJLJ_SETJMP(Op, DAG);
10340 case ISD::EH_SJLJ_LONGJMP: return LowerEH_SJLJ_LONGJMP(Op, DAG);
10341 case ISD::EH_SJLJ_SETUP_DISPATCH: return LowerEH_SJLJ_SETUP_DISPATCH(Op, DAG);
10342 case ISD::INTRINSIC_VOID: return LowerINTRINSIC_VOID(Op, DAG, Subtarget);
10343 case ISD::INTRINSIC_WO_CHAIN: return LowerINTRINSIC_WO_CHAIN(Op, DAG,
10344 Subtarget);
10345 case ISD::BITCAST: return ExpandBITCAST(Op.getNode(), DAG, Subtarget);
10346 case ISD::SHL:
10347 case ISD::SRL:
10348 case ISD::SRA: return LowerShift(Op.getNode(), DAG, Subtarget);
10349 case ISD::SREM: return LowerREM(Op.getNode(), DAG);
10350 case ISD::UREM: return LowerREM(Op.getNode(), DAG);
10351 case ISD::SHL_PARTS: return LowerShiftLeftParts(Op, DAG);
10352 case ISD::SRL_PARTS:
10353 case ISD::SRA_PARTS: return LowerShiftRightParts(Op, DAG);
10354 case ISD::CTTZ:
10355 case ISD::CTTZ_ZERO_POISON: return LowerCTTZ(Op.getNode(), DAG, Subtarget);
10356 case ISD::CTPOP: return LowerCTPOP(Op.getNode(), DAG, Subtarget);
10357 case ISD::SETCC: return LowerVSETCC(Op, DAG, Subtarget);
10358 case ISD::SETCCCARRY: return LowerSETCCCARRY(Op, DAG);
10359 case ISD::ConstantFP: return LowerConstantFP(Op, DAG, Subtarget);
10360 case ISD::BUILD_VECTOR: return LowerBUILD_VECTOR(Op, DAG, Subtarget);
10361 case ISD::VECTOR_SHUFFLE: return LowerVECTOR_SHUFFLE(Op, DAG, Subtarget);
10362 case ISD::EXTRACT_SUBVECTOR: return LowerEXTRACT_SUBVECTOR(Op, DAG, Subtarget);
10363 case ISD::INSERT_VECTOR_ELT: return LowerINSERT_VECTOR_ELT(Op, DAG);
10364 case ISD::EXTRACT_VECTOR_ELT: return LowerEXTRACT_VECTOR_ELT(Op, DAG, Subtarget);
10365 case ISD::CONCAT_VECTORS: return LowerCONCAT_VECTORS(Op, DAG, Subtarget);
10366 case ISD::TRUNCATE: return LowerTruncate(Op.getNode(), DAG, Subtarget);
10367 case ISD::SIGN_EXTEND:
10368 case ISD::ZERO_EXTEND: return LowerVectorExtend(Op.getNode(), DAG, Subtarget);
10369 case ISD::GET_ROUNDING: return LowerGET_ROUNDING(Op, DAG);
10370 case ISD::SET_ROUNDING: return LowerSET_ROUNDING(Op, DAG);
10371 case ISD::SET_FPMODE:
10372 return LowerSET_FPMODE(Op, DAG);
10373 case ISD::RESET_FPMODE:
10374 return LowerRESET_FPMODE(Op, DAG);
10375 case ISD::MUL: return LowerMUL(Op, DAG);
10376 case ISD::SDIV:
10377 if (getTargetMachine().getTargetTriple().isOSWindows() &&
10378 !Op.getValueType().isVector())
10379 return LowerDIV_Windows(Op, DAG, /* Signed */ true);
10380 return LowerSDIV(Op, DAG, Subtarget);
10381 case ISD::UDIV:
10382 if (getTargetMachine().getTargetTriple().isOSWindows() &&
10383 !Op.getValueType().isVector())
10384 return LowerDIV_Windows(Op, DAG, /* Signed */ false);
10385 return LowerUDIV(Op, DAG, Subtarget);
10386 case ISD::UADDO_CARRY:
10387 return LowerADDSUBO_CARRY(Op, DAG, ARMISD::ADDE, false /*unsigned*/);
10388 case ISD::USUBO_CARRY:
10389 return LowerADDSUBO_CARRY(Op, DAG, ARMISD::SUBE, false /*unsigned*/);
10390 case ISD::SADDO_CARRY:
10391 return LowerADDSUBO_CARRY(Op, DAG, ARMISD::ADDE, true /*signed*/);
10392 case ISD::SSUBO_CARRY:
10393 return LowerADDSUBO_CARRY(Op, DAG, ARMISD::SUBE, true /*signed*/);
10394 case ISD::UADDO:
10395 case ISD::USUBO:
10396 case ISD::UMULO:
10397 case ISD::SADDO:
10398 case ISD::SSUBO:
10399 case ISD::SMULO:
10400 return LowerALUO(Op, DAG);
10401 case ISD::SADDSAT:
10402 case ISD::SSUBSAT:
10403 case ISD::UADDSAT:
10404 case ISD::USUBSAT:
10405 return LowerADDSUBSAT(Op, DAG, Subtarget);
10406 case ISD::LOAD: {
10407 auto *LD = cast<LoadSDNode>(Op);
10408 EVT MemVT = LD->getMemoryVT();
10409 if (Subtarget->hasMVEIntegerOps() &&
10410 (MemVT == MVT::v2i1 || MemVT == MVT::v4i1 || MemVT == MVT::v8i1 ||
10411 MemVT == MVT::v16i1))
10412 return LowerPredicateLoad(Op, DAG);
10413
10414 auto Pair = LowerAEABIUnalignedLoad(Op, DAG);
10415 if (Pair.first)
10416 return DAG.getMergeValues({Pair.first, Pair.second}, SDLoc(Pair.first));
10417 return SDValue();
10418 }
10419 case ISD::STORE:
10420 return LowerSTORE(Op, DAG, Subtarget);
10421 case ISD::MLOAD:
10422 return LowerMLOAD(Op, DAG);
10423 case ISD::VECREDUCE_MUL:
10424 case ISD::VECREDUCE_AND:
10425 case ISD::VECREDUCE_OR:
10426 case ISD::VECREDUCE_XOR:
10427 return LowerVecReduce(Op, DAG, Subtarget);
10432 return LowerVecReduceF(Op, DAG, Subtarget);
10437 return LowerVecReduceMinMax(Op, DAG, Subtarget);
10438 case ISD::ATOMIC_LOAD:
10439 case ISD::ATOMIC_STORE:
10440 return LowerAtomicLoadStore(Op, DAG);
10441 case ISD::SDIVREM:
10442 case ISD::UDIVREM: return LowerDivRem(Op, DAG);
10444 if (getTargetMachine().getTargetTriple().isOSWindows())
10445 return LowerDYNAMIC_STACKALLOC(Op, DAG);
10446 llvm_unreachable("Don't know how to custom lower this!");
10448 case ISD::FP_ROUND: return LowerFP_ROUND(Op, DAG);
10450 case ISD::FP_EXTEND: return LowerFP_EXTEND(Op, DAG);
10451 case ISD::STRICT_FSETCC:
10452 case ISD::STRICT_FSETCCS: return LowerFSETCC(Op, DAG);
10453 case ISD::SPONENTRY:
10454 return LowerSPONENTRY(Op, DAG);
10455 case ISD::FP_TO_BF16:
10456 return LowerFP_TO_BF16(Op, DAG);
10457 case ARMISD::WIN__DBZCHK: return SDValue();
10458 case ISD::UCMP:
10459 case ISD::SCMP:
10460 return LowerCMP(Op, DAG);
10461 case ISD::ABS:
10462 return LowerABS(Op, DAG);
10463 case ISD::STRICT_LROUND:
10465 case ISD::STRICT_LRINT:
10466 case ISD::STRICT_LLRINT: {
10467 assert((Op.getOperand(1).getValueType() == MVT::f16 ||
10468 Op.getOperand(1).getValueType() == MVT::bf16) &&
10469 "Expected custom lowering of rounding operations only for f16");
10470 SDLoc DL(Op);
10471 SDValue Ext = DAG.getNode(ISD::STRICT_FP_EXTEND, DL, {MVT::f32, MVT::Other},
10472 {Op.getOperand(0), Op.getOperand(1)});
10473 return DAG.getNode(Op.getOpcode(), DL, {Op.getValueType(), MVT::Other},
10474 {Ext.getValue(1), Ext.getValue(0)});
10475 }
10476 }
10477}
10478
10480 SelectionDAG &DAG) {
10481 unsigned IntNo = N->getConstantOperandVal(0);
10482 unsigned Opc = 0;
10483 if (IntNo == Intrinsic::arm_smlald)
10484 Opc = ARMISD::SMLALD;
10485 else if (IntNo == Intrinsic::arm_smlaldx)
10486 Opc = ARMISD::SMLALDX;
10487 else if (IntNo == Intrinsic::arm_smlsld)
10488 Opc = ARMISD::SMLSLD;
10489 else if (IntNo == Intrinsic::arm_smlsldx)
10490 Opc = ARMISD::SMLSLDX;
10491 else
10492 return;
10493
10494 SDLoc dl(N);
10495 SDValue Lo, Hi;
10496 std::tie(Lo, Hi) = DAG.SplitScalar(N->getOperand(3), dl, MVT::i32, MVT::i32);
10497
10498 SDValue LongMul = DAG.getNode(Opc, dl,
10499 DAG.getVTList(MVT::i32, MVT::i32),
10500 N->getOperand(1), N->getOperand(2),
10501 Lo, Hi);
10502 Results.push_back(DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64,
10503 LongMul.getValue(0), LongMul.getValue(1)));
10504}
10505
10506/// ReplaceNodeResults - Replace the results of node with an illegal result
10507/// type with new values built out of custom code.
10510 SelectionDAG &DAG) const {
10511 SDValue Res;
10512 switch (N->getOpcode()) {
10513 default:
10514 llvm_unreachable("Don't know how to custom expand this!");
10515 case ISD::READ_REGISTER:
10517 break;
10518 case ISD::BITCAST:
10519 Res = ExpandBITCAST(N, DAG, Subtarget);
10520 break;
10521 case ISD::SRL:
10522 case ISD::SRA:
10523 case ISD::SHL:
10524 Res = Expand64BitShift(N, DAG, Subtarget);
10525 break;
10526 case ISD::SREM:
10527 case ISD::UREM:
10528 Res = LowerREM(N, DAG);
10529 break;
10530 case ISD::SDIVREM:
10531 case ISD::UDIVREM:
10532 Res = LowerDivRem(SDValue(N, 0), DAG);
10533 assert(Res.getNumOperands() == 2 && "DivRem needs two values");
10534 Results.push_back(Res.getValue(0));
10535 Results.push_back(Res.getValue(1));
10536 return;
10537 case ISD::SADDSAT:
10538 case ISD::SSUBSAT:
10539 case ISD::UADDSAT:
10540 case ISD::USUBSAT:
10541 Res = LowerADDSUBSAT(SDValue(N, 0), DAG, Subtarget);
10542 break;
10544 ReplaceREADCYCLECOUNTER(N, Results, DAG, Subtarget);
10545 return;
10546 case ISD::UDIV:
10547 case ISD::SDIV:
10548 assert(getTargetMachine().getTargetTriple().isOSWindows() &&
10549 "can only expand DIV on Windows");
10550 return ExpandDIV_Windows(SDValue(N, 0), DAG, N->getOpcode() == ISD::SDIV,
10551 Results);
10554 return;
10556 return ReplaceLongIntrinsic(N, Results, DAG);
10557 case ISD::LOAD:
10558 LowerLOAD(N, Results, DAG);
10559 break;
10560 case ISD::STORE:
10561 Res = LowerAEABIUnalignedStore(SDValue(N, 0), DAG);
10562 break;
10563 case ISD::TRUNCATE:
10564 Res = LowerTruncate(N, DAG, Subtarget);
10565 break;
10566 case ISD::SIGN_EXTEND:
10567 case ISD::ZERO_EXTEND:
10568 Res = LowerVectorExtend(N, DAG, Subtarget);
10569 break;
10572 Res = LowerFP_TO_INT_SAT(SDValue(N, 0), DAG, Subtarget);
10573 break;
10574 }
10575 if (Res.getNode())
10576 Results.push_back(Res);
10577}
10578
10579//===----------------------------------------------------------------------===//
10580// ARM Scheduler Hooks
10581//===----------------------------------------------------------------------===//
10582
10583/// SetupEntryBlockForSjLj - Insert code into the entry block that creates and
10584/// registers the function context.
10585void ARMTargetLowering::SetupEntryBlockForSjLj(MachineInstr &MI,
10587 MachineBasicBlock *DispatchBB,
10588 int FI) const {
10589 assert(!Subtarget->isROPI() && !Subtarget->isRWPI() &&
10590 "ROPI/RWPI not currently supported with SjLj");
10591 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
10592 DebugLoc dl = MI.getDebugLoc();
10593 MachineFunction *MF = MBB->getParent();
10594 MachineRegisterInfo *MRI = &MF->getRegInfo();
10597 const Function &F = MF->getFunction();
10598
10599 bool isThumb = Subtarget->isThumb();
10600 bool isThumb2 = Subtarget->isThumb2();
10601
10602 unsigned PCLabelId = AFI->createPICLabelUId();
10603 unsigned PCAdj = (isThumb || isThumb2) ? 4 : 8;
10605 ARMConstantPoolMBB::Create(F.getContext(), DispatchBB, PCLabelId, PCAdj);
10606 unsigned CPI = MCP->getConstantPoolIndex(CPV, Align(4));
10607
10608 const TargetRegisterClass *TRC = isThumb ? &ARM::tGPRRegClass
10609 : &ARM::GPRRegClass;
10610
10611 // Grab constant pool and fixed stack memory operands.
10612 MachineMemOperand *CPMMO =
10615
10616 MachineMemOperand *FIMMOSt =
10619
10620 // Load the address of the dispatch MBB into the jump buffer.
10621 if (isThumb2) {
10622 // Incoming value: jbuf
10623 // ldr.n r5, LCPI1_1
10624 // orr r5, r5, #1
10625 // add r5, pc
10626 // str r5, [$jbuf, #+4] ; &jbuf[1]
10627 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10628 BuildMI(*MBB, MI, dl, TII->get(ARM::t2LDRpci), NewVReg1)
10630 .addMemOperand(CPMMO)
10632 // Set the low bit because of thumb mode.
10633 Register NewVReg2 = MRI->createVirtualRegister(TRC);
10634 BuildMI(*MBB, MI, dl, TII->get(ARM::t2ORRri), NewVReg2)
10635 .addReg(NewVReg1)
10636 .addImm(0x01)
10638 .add(condCodeOp());
10639 Register NewVReg3 = MRI->createVirtualRegister(TRC);
10640 BuildMI(*MBB, MI, dl, TII->get(ARM::tPICADD), NewVReg3)
10641 .addReg(NewVReg2)
10642 .addImm(PCLabelId);
10643 BuildMI(*MBB, MI, dl, TII->get(ARM::t2STRi12))
10644 .addReg(NewVReg3)
10645 .addFrameIndex(FI)
10646 .addImm(36) // &jbuf[1] :: pc
10647 .addMemOperand(FIMMOSt)
10649 } else if (isThumb) {
10650 // Incoming value: jbuf
10651 // ldr.n r1, LCPI1_4
10652 // add r1, pc
10653 // mov r2, #1
10654 // orrs r1, r2
10655 // add r2, $jbuf, #+4 ; &jbuf[1]
10656 // str r1, [r2]
10657 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10658 BuildMI(*MBB, MI, dl, TII->get(ARM::tLDRpci), NewVReg1)
10660 .addMemOperand(CPMMO)
10662 Register NewVReg2 = MRI->createVirtualRegister(TRC);
10663 BuildMI(*MBB, MI, dl, TII->get(ARM::tPICADD), NewVReg2)
10664 .addReg(NewVReg1)
10665 .addImm(PCLabelId);
10666 // Set the low bit because of thumb mode.
10667 Register NewVReg3 = MRI->createVirtualRegister(TRC);
10668 BuildMI(*MBB, MI, dl, TII->get(ARM::tMOVi8), NewVReg3)
10669 .addReg(ARM::CPSR, RegState::Define)
10670 .addImm(1)
10672 Register NewVReg4 = MRI->createVirtualRegister(TRC);
10673 BuildMI(*MBB, MI, dl, TII->get(ARM::tORR), NewVReg4)
10674 .addReg(ARM::CPSR, RegState::Define)
10675 .addReg(NewVReg2)
10676 .addReg(NewVReg3)
10678 Register NewVReg5 = MRI->createVirtualRegister(TRC);
10679 BuildMI(*MBB, MI, dl, TII->get(ARM::tADDframe), NewVReg5)
10680 .addFrameIndex(FI)
10681 .addImm(36); // &jbuf[1] :: pc
10682 BuildMI(*MBB, MI, dl, TII->get(ARM::tSTRi))
10683 .addReg(NewVReg4)
10684 .addReg(NewVReg5)
10685 .addImm(0)
10686 .addMemOperand(FIMMOSt)
10688 } else {
10689 // Incoming value: jbuf
10690 // ldr r1, LCPI1_1
10691 // add r1, pc, r1
10692 // str r1, [$jbuf, #+4] ; &jbuf[1]
10693 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10694 BuildMI(*MBB, MI, dl, TII->get(ARM::LDRi12), NewVReg1)
10696 .addImm(0)
10697 .addMemOperand(CPMMO)
10699 Register NewVReg2 = MRI->createVirtualRegister(TRC);
10700 BuildMI(*MBB, MI, dl, TII->get(ARM::PICADD), NewVReg2)
10701 .addReg(NewVReg1)
10702 .addImm(PCLabelId)
10704 BuildMI(*MBB, MI, dl, TII->get(ARM::STRi12))
10705 .addReg(NewVReg2)
10706 .addFrameIndex(FI)
10707 .addImm(36) // &jbuf[1] :: pc
10708 .addMemOperand(FIMMOSt)
10710 }
10711}
10712
10713void ARMTargetLowering::EmitSjLjDispatchBlock(MachineInstr &MI,
10714 MachineBasicBlock *MBB) const {
10715 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
10716 DebugLoc dl = MI.getDebugLoc();
10717 MachineFunction *MF = MBB->getParent();
10718 MachineRegisterInfo *MRI = &MF->getRegInfo();
10719 MachineFrameInfo &MFI = MF->getFrameInfo();
10720 int FI = MFI.getFunctionContextIndex();
10721
10722 const TargetRegisterClass *TRC = Subtarget->isThumb() ? &ARM::tGPRRegClass
10723 : &ARM::GPRnopcRegClass;
10724
10725 // Get a mapping of the call site numbers to all of the landing pads they're
10726 // associated with.
10727 DenseMap<unsigned, SmallVector<MachineBasicBlock*, 2>> CallSiteNumToLPad;
10728 unsigned MaxCSNum = 0;
10729 for (MachineBasicBlock &BB : *MF) {
10730 if (!BB.isEHPad())
10731 continue;
10732
10733 // FIXME: We should assert that the EH_LABEL is the first MI in the landing
10734 // pad.
10735 for (MachineInstr &II : BB) {
10736 if (!II.isEHLabel())
10737 continue;
10738
10739 MCSymbol *Sym = II.getOperand(0).getMCSymbol();
10740 if (!MF->hasCallSiteLandingPad(Sym)) continue;
10741
10742 SmallVectorImpl<unsigned> &CallSiteIdxs = MF->getCallSiteLandingPad(Sym);
10743 for (unsigned Idx : CallSiteIdxs) {
10744 CallSiteNumToLPad[Idx].push_back(&BB);
10745 MaxCSNum = std::max(MaxCSNum, Idx);
10746 }
10747 break;
10748 }
10749 }
10750
10751 // Get an ordered list of the machine basic blocks for the jump table.
10752 std::vector<MachineBasicBlock*> LPadList;
10753 SmallPtrSet<MachineBasicBlock*, 32> InvokeBBs;
10754 LPadList.reserve(CallSiteNumToLPad.size());
10755 for (unsigned I = 1; I <= MaxCSNum; ++I) {
10756 SmallVectorImpl<MachineBasicBlock*> &MBBList = CallSiteNumToLPad[I];
10757 for (MachineBasicBlock *MBB : MBBList) {
10758 LPadList.push_back(MBB);
10759 InvokeBBs.insert_range(MBB->predecessors());
10760 }
10761 }
10762
10763 assert(!LPadList.empty() &&
10764 "No landing pad destinations for the dispatch jump table!");
10765
10766 // Create the jump table and associated information.
10767 MachineJumpTableInfo *JTI =
10768 MF->getOrCreateJumpTableInfo(MachineJumpTableInfo::EK_Inline);
10769 unsigned MJTI = JTI->createJumpTableIndex(LPadList);
10770
10771 // Create the MBBs for the dispatch code.
10772
10773 // Shove the dispatch's address into the return slot in the function context.
10774 MachineBasicBlock *DispatchBB = MF->CreateMachineBasicBlock();
10775 DispatchBB->setIsEHPad();
10776
10777 MachineBasicBlock *TrapBB = MF->CreateMachineBasicBlock();
10778
10779 BuildMI(TrapBB, dl, TII->get(Subtarget->isThumb() ? ARM::tTRAP : ARM::TRAP));
10780 DispatchBB->addSuccessor(TrapBB);
10781
10782 MachineBasicBlock *DispContBB = MF->CreateMachineBasicBlock();
10783 DispatchBB->addSuccessor(DispContBB);
10784
10785 // Insert and MBBs.
10786 MF->insert(MF->end(), DispatchBB);
10787 MF->insert(MF->end(), DispContBB);
10788 MF->insert(MF->end(), TrapBB);
10789
10790 // Insert code into the entry block that creates and registers the function
10791 // context.
10792 SetupEntryBlockForSjLj(MI, MBB, DispatchBB, FI);
10793
10794 MachineMemOperand *FIMMOLd = MF->getMachineMemOperand(
10797
10798 MachineInstrBuilder MIB;
10799 MIB = BuildMI(DispatchBB, dl, TII->get(ARM::Int_eh_sjlj_dispatchsetup));
10800
10801 const ARMBaseInstrInfo *AII = static_cast<const ARMBaseInstrInfo*>(TII);
10802 const ARMBaseRegisterInfo &RI = AII->getRegisterInfo();
10803
10804 // Add a register mask with no preserved registers. This results in all
10805 // registers being marked as clobbered. This can't work if the dispatch block
10806 // is in a Thumb1 function and is linked with ARM code which uses the FP
10807 // registers, as there is no way to preserve the FP registers in Thumb1 mode.
10809
10810 bool IsPositionIndependent = isPositionIndependent();
10811 unsigned NumLPads = LPadList.size();
10812 if (Subtarget->isThumb2()) {
10813 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10814 BuildMI(DispatchBB, dl, TII->get(ARM::t2LDRi12), NewVReg1)
10815 .addFrameIndex(FI)
10816 .addImm(4)
10817 .addMemOperand(FIMMOLd)
10819
10820 if (NumLPads < 256) {
10821 BuildMI(DispatchBB, dl, TII->get(ARM::t2CMPri))
10822 .addReg(NewVReg1)
10823 .addImm(LPadList.size())
10825 } else {
10826 Register VReg1 = MRI->createVirtualRegister(TRC);
10827 BuildMI(DispatchBB, dl, TII->get(ARM::t2MOVi16), VReg1)
10828 .addImm(NumLPads & 0xFFFF)
10830
10831 unsigned VReg2 = VReg1;
10832 if ((NumLPads & 0xFFFF0000) != 0) {
10833 VReg2 = MRI->createVirtualRegister(TRC);
10834 BuildMI(DispatchBB, dl, TII->get(ARM::t2MOVTi16), VReg2)
10835 .addReg(VReg1)
10836 .addImm(NumLPads >> 16)
10838 }
10839
10840 BuildMI(DispatchBB, dl, TII->get(ARM::t2CMPrr))
10841 .addReg(NewVReg1)
10842 .addReg(VReg2)
10844 }
10845
10846 BuildMI(DispatchBB, dl, TII->get(ARM::t2Bcc))
10847 .addMBB(TrapBB)
10849 .addReg(ARM::CPSR);
10850
10851 Register NewVReg3 = MRI->createVirtualRegister(TRC);
10852 BuildMI(DispContBB, dl, TII->get(ARM::t2LEApcrelJT), NewVReg3)
10853 .addJumpTableIndex(MJTI)
10855
10856 Register NewVReg4 = MRI->createVirtualRegister(TRC);
10857 BuildMI(DispContBB, dl, TII->get(ARM::t2ADDrs), NewVReg4)
10858 .addReg(NewVReg3)
10859 .addReg(NewVReg1)
10862 .add(condCodeOp());
10863
10864 BuildMI(DispContBB, dl, TII->get(ARM::t2BR_JT))
10865 .addReg(NewVReg4)
10866 .addReg(NewVReg1)
10867 .addJumpTableIndex(MJTI);
10868 } else if (Subtarget->isThumb()) {
10869 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10870 BuildMI(DispatchBB, dl, TII->get(ARM::tLDRspi), NewVReg1)
10871 .addFrameIndex(FI)
10872 .addImm(1)
10873 .addMemOperand(FIMMOLd)
10875
10876 if (NumLPads < 256) {
10877 BuildMI(DispatchBB, dl, TII->get(ARM::tCMPi8))
10878 .addReg(NewVReg1)
10879 .addImm(NumLPads)
10881 } else {
10882 MachineConstantPool *ConstantPool = MF->getConstantPool();
10883 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext());
10884 const Constant *C = ConstantInt::get(Int32Ty, NumLPads);
10885
10886 // MachineConstantPool wants an explicit alignment.
10887 Align Alignment = MF->getDataLayout().getPrefTypeAlign(Int32Ty);
10888 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Alignment);
10889
10890 Register VReg1 = MRI->createVirtualRegister(TRC);
10891 BuildMI(DispatchBB, dl, TII->get(ARM::tLDRpci))
10892 .addReg(VReg1, RegState::Define)
10895 BuildMI(DispatchBB, dl, TII->get(ARM::tCMPr))
10896 .addReg(NewVReg1)
10897 .addReg(VReg1)
10899 }
10900
10901 BuildMI(DispatchBB, dl, TII->get(ARM::tBcc))
10902 .addMBB(TrapBB)
10904 .addReg(ARM::CPSR);
10905
10906 Register NewVReg2 = MRI->createVirtualRegister(TRC);
10907 BuildMI(DispContBB, dl, TII->get(ARM::tLSLri), NewVReg2)
10908 .addReg(ARM::CPSR, RegState::Define)
10909 .addReg(NewVReg1)
10910 .addImm(2)
10912
10913 Register NewVReg3 = MRI->createVirtualRegister(TRC);
10914 BuildMI(DispContBB, dl, TII->get(ARM::tLEApcrelJT), NewVReg3)
10915 .addJumpTableIndex(MJTI)
10917
10918 Register NewVReg4 = MRI->createVirtualRegister(TRC);
10919 BuildMI(DispContBB, dl, TII->get(ARM::tADDrr), NewVReg4)
10920 .addReg(ARM::CPSR, RegState::Define)
10921 .addReg(NewVReg2)
10922 .addReg(NewVReg3)
10924
10925 MachineMemOperand *JTMMOLd =
10926 MF->getMachineMemOperand(MachinePointerInfo::getJumpTable(*MF),
10928
10929 Register NewVReg5 = MRI->createVirtualRegister(TRC);
10930 BuildMI(DispContBB, dl, TII->get(ARM::tLDRi), NewVReg5)
10931 .addReg(NewVReg4)
10932 .addImm(0)
10933 .addMemOperand(JTMMOLd)
10935
10936 unsigned NewVReg6 = NewVReg5;
10937 if (IsPositionIndependent) {
10938 NewVReg6 = MRI->createVirtualRegister(TRC);
10939 BuildMI(DispContBB, dl, TII->get(ARM::tADDrr), NewVReg6)
10940 .addReg(ARM::CPSR, RegState::Define)
10941 .addReg(NewVReg5)
10942 .addReg(NewVReg3)
10944 }
10945
10946 BuildMI(DispContBB, dl, TII->get(ARM::tBR_JTr))
10947 .addReg(NewVReg6)
10948 .addJumpTableIndex(MJTI);
10949 } else {
10950 Register NewVReg1 = MRI->createVirtualRegister(TRC);
10951 BuildMI(DispatchBB, dl, TII->get(ARM::LDRi12), NewVReg1)
10952 .addFrameIndex(FI)
10953 .addImm(4)
10954 .addMemOperand(FIMMOLd)
10956
10957 if (NumLPads < 256) {
10958 BuildMI(DispatchBB, dl, TII->get(ARM::CMPri))
10959 .addReg(NewVReg1)
10960 .addImm(NumLPads)
10962 } else if (Subtarget->hasV6T2Ops() && isUInt<16>(NumLPads)) {
10963 Register VReg1 = MRI->createVirtualRegister(TRC);
10964 BuildMI(DispatchBB, dl, TII->get(ARM::MOVi16), VReg1)
10965 .addImm(NumLPads & 0xFFFF)
10967
10968 unsigned VReg2 = VReg1;
10969 if ((NumLPads & 0xFFFF0000) != 0) {
10970 VReg2 = MRI->createVirtualRegister(TRC);
10971 BuildMI(DispatchBB, dl, TII->get(ARM::MOVTi16), VReg2)
10972 .addReg(VReg1)
10973 .addImm(NumLPads >> 16)
10975 }
10976
10977 BuildMI(DispatchBB, dl, TII->get(ARM::CMPrr))
10978 .addReg(NewVReg1)
10979 .addReg(VReg2)
10981 } else {
10982 MachineConstantPool *ConstantPool = MF->getConstantPool();
10983 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext());
10984 const Constant *C = ConstantInt::get(Int32Ty, NumLPads);
10985
10986 // MachineConstantPool wants an explicit alignment.
10987 Align Alignment = MF->getDataLayout().getPrefTypeAlign(Int32Ty);
10988 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Alignment);
10989
10990 Register VReg1 = MRI->createVirtualRegister(TRC);
10991 BuildMI(DispatchBB, dl, TII->get(ARM::LDRcp))
10992 .addReg(VReg1, RegState::Define)
10994 .addImm(0)
10996 BuildMI(DispatchBB, dl, TII->get(ARM::CMPrr))
10997 .addReg(NewVReg1)
10998 .addReg(VReg1)
11000 }
11001
11002 BuildMI(DispatchBB, dl, TII->get(ARM::Bcc))
11003 .addMBB(TrapBB)
11005 .addReg(ARM::CPSR);
11006
11007 Register NewVReg3 = MRI->createVirtualRegister(TRC);
11008 BuildMI(DispContBB, dl, TII->get(ARM::MOVsi), NewVReg3)
11009 .addReg(NewVReg1)
11012 .add(condCodeOp());
11013 Register NewVReg4 = MRI->createVirtualRegister(TRC);
11014 BuildMI(DispContBB, dl, TII->get(ARM::LEApcrelJT), NewVReg4)
11015 .addJumpTableIndex(MJTI)
11017
11018 MachineMemOperand *JTMMOLd =
11019 MF->getMachineMemOperand(MachinePointerInfo::getJumpTable(*MF),
11021 Register NewVReg5 = MRI->createVirtualRegister(TRC);
11022 BuildMI(DispContBB, dl, TII->get(ARM::LDRrs), NewVReg5)
11023 .addReg(NewVReg3)
11024 .addReg(NewVReg4)
11025 .addImm(0)
11026 .addMemOperand(JTMMOLd)
11028
11029 if (IsPositionIndependent) {
11030 BuildMI(DispContBB, dl, TII->get(ARM::BR_JTadd))
11031 .addReg(NewVReg5)
11032 .addReg(NewVReg4)
11033 .addJumpTableIndex(MJTI);
11034 } else {
11035 BuildMI(DispContBB, dl, TII->get(ARM::BR_JTr))
11036 .addReg(NewVReg5)
11037 .addJumpTableIndex(MJTI);
11038 }
11039 }
11040
11041 // Add the jump table entries as successors to the MBB.
11042 SmallPtrSet<MachineBasicBlock*, 8> SeenMBBs;
11043 for (MachineBasicBlock *CurMBB : LPadList) {
11044 if (SeenMBBs.insert(CurMBB).second)
11045 DispContBB->addSuccessor(CurMBB);
11046 }
11047
11048 // N.B. the order the invoke BBs are processed in doesn't matter here.
11049 const MCPhysReg *SavedRegs = RI.getCalleeSavedRegs(MF);
11051 for (MachineBasicBlock *BB : InvokeBBs) {
11052
11053 // Remove the landing pad successor from the invoke block and replace it
11054 // with the new dispatch block.
11055 SmallVector<MachineBasicBlock*, 4> Successors(BB->successors());
11056 while (!Successors.empty()) {
11057 MachineBasicBlock *SMBB = Successors.pop_back_val();
11058 if (SMBB->isEHPad()) {
11059 BB->removeSuccessor(SMBB);
11060 MBBLPads.push_back(SMBB);
11061 }
11062 }
11063
11064 BB->addSuccessor(DispatchBB, BranchProbability::getZero());
11065 BB->normalizeSuccProbs();
11066
11067 // Find the invoke call and mark all of the callee-saved registers as
11068 // 'implicit defined' so that they're spilled. This prevents code from
11069 // moving instructions to before the EH block, where they will never be
11070 // executed.
11072 II = BB->rbegin(), IE = BB->rend(); II != IE; ++II) {
11073 if (!II->isCall()) continue;
11074
11075 DenseSet<unsigned> DefRegs;
11077 OI = II->operands_begin(), OE = II->operands_end();
11078 OI != OE; ++OI) {
11079 if (!OI->isReg()) continue;
11080 DefRegs.insert(OI->getReg());
11081 }
11082
11083 MachineInstrBuilder MIB(*MF, &*II);
11084
11085 for (unsigned i = 0; SavedRegs[i] != 0; ++i) {
11086 unsigned Reg = SavedRegs[i];
11087 if (Subtarget->isThumb2() &&
11088 !ARM::tGPRRegClass.contains(Reg) &&
11089 !ARM::hGPRRegClass.contains(Reg))
11090 continue;
11091 if (Subtarget->isThumb1Only() && !ARM::tGPRRegClass.contains(Reg))
11092 continue;
11093 if (!Subtarget->isThumb() && !ARM::GPRRegClass.contains(Reg))
11094 continue;
11095 if (!DefRegs.contains(Reg))
11097 }
11098
11099 break;
11100 }
11101 }
11102
11103 // Mark all former landing pads as non-landing pads. The dispatch is the only
11104 // landing pad now.
11105 for (MachineBasicBlock *MBBLPad : MBBLPads)
11106 MBBLPad->setIsEHPad(false);
11107
11108 // The instruction is gone now.
11109 MI.eraseFromParent();
11110}
11111
11112static
11114 for (MachineBasicBlock *S : MBB->successors())
11115 if (S != Succ)
11116 return S;
11117 llvm_unreachable("Expecting a BB with two successors!");
11118}
11119
11120/// Return the load opcode for a given load size. If load size >= 8,
11121/// neon opcode will be returned.
11122static unsigned getLdOpcode(unsigned LdSize, bool IsThumb1, bool IsThumb2) {
11123 if (LdSize >= 8)
11124 return LdSize == 16 ? ARM::VLD1q32wb_fixed
11125 : LdSize == 8 ? ARM::VLD1d32wb_fixed : 0;
11126 if (IsThumb1)
11127 return LdSize == 4 ? ARM::tLDRi
11128 : LdSize == 2 ? ARM::tLDRHi
11129 : LdSize == 1 ? ARM::tLDRBi : 0;
11130 if (IsThumb2)
11131 return LdSize == 4 ? ARM::t2LDR_POST
11132 : LdSize == 2 ? ARM::t2LDRH_POST
11133 : LdSize == 1 ? ARM::t2LDRB_POST : 0;
11134 return LdSize == 4 ? ARM::LDR_POST_IMM
11135 : LdSize == 2 ? ARM::LDRH_POST
11136 : LdSize == 1 ? ARM::LDRB_POST_IMM : 0;
11137}
11138
11139/// Return the store opcode for a given store size. If store size >= 8,
11140/// neon opcode will be returned.
11141static unsigned getStOpcode(unsigned StSize, bool IsThumb1, bool IsThumb2) {
11142 if (StSize >= 8)
11143 return StSize == 16 ? ARM::VST1q32wb_fixed
11144 : StSize == 8 ? ARM::VST1d32wb_fixed : 0;
11145 if (IsThumb1)
11146 return StSize == 4 ? ARM::tSTRi
11147 : StSize == 2 ? ARM::tSTRHi
11148 : StSize == 1 ? ARM::tSTRBi : 0;
11149 if (IsThumb2)
11150 return StSize == 4 ? ARM::t2STR_POST
11151 : StSize == 2 ? ARM::t2STRH_POST
11152 : StSize == 1 ? ARM::t2STRB_POST : 0;
11153 return StSize == 4 ? ARM::STR_POST_IMM
11154 : StSize == 2 ? ARM::STRH_POST
11155 : StSize == 1 ? ARM::STRB_POST_IMM : 0;
11156}
11157
11158/// Emit a post-increment load operation with given size. The instructions
11159/// will be added to BB at Pos.
11161 const TargetInstrInfo *TII, const DebugLoc &dl,
11162 unsigned LdSize, unsigned Data, unsigned AddrIn,
11163 unsigned AddrOut, bool IsThumb1, bool IsThumb2) {
11164 unsigned LdOpc = getLdOpcode(LdSize, IsThumb1, IsThumb2);
11165 assert(LdOpc != 0 && "Should have a load opcode");
11166 if (LdSize >= 8) {
11167 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data)
11168 .addReg(AddrOut, RegState::Define)
11169 .addReg(AddrIn)
11170 .addImm(0)
11172 } else if (IsThumb1) {
11173 // load + update AddrIn
11174 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data)
11175 .addReg(AddrIn)
11176 .addImm(0)
11178 BuildMI(*BB, Pos, dl, TII->get(ARM::tADDi8), AddrOut)
11179 .add(t1CondCodeOp())
11180 .addReg(AddrIn)
11181 .addImm(LdSize)
11183 } else if (IsThumb2) {
11184 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data)
11185 .addReg(AddrOut, RegState::Define)
11186 .addReg(AddrIn)
11187 .addImm(LdSize)
11189 } else { // arm
11190 BuildMI(*BB, Pos, dl, TII->get(LdOpc), Data)
11191 .addReg(AddrOut, RegState::Define)
11192 .addReg(AddrIn)
11193 .addReg(0)
11194 .addImm(LdSize)
11196 }
11197}
11198
11199/// Emit a post-increment store operation with given size. The instructions
11200/// will be added to BB at Pos.
11202 const TargetInstrInfo *TII, const DebugLoc &dl,
11203 unsigned StSize, unsigned Data, unsigned AddrIn,
11204 unsigned AddrOut, bool IsThumb1, bool IsThumb2) {
11205 unsigned StOpc = getStOpcode(StSize, IsThumb1, IsThumb2);
11206 assert(StOpc != 0 && "Should have a store opcode");
11207 if (StSize >= 8) {
11208 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut)
11209 .addReg(AddrIn)
11210 .addImm(0)
11211 .addReg(Data)
11213 } else if (IsThumb1) {
11214 // store + update AddrIn
11215 BuildMI(*BB, Pos, dl, TII->get(StOpc))
11216 .addReg(Data)
11217 .addReg(AddrIn)
11218 .addImm(0)
11220 BuildMI(*BB, Pos, dl, TII->get(ARM::tADDi8), AddrOut)
11221 .add(t1CondCodeOp())
11222 .addReg(AddrIn)
11223 .addImm(StSize)
11225 } else if (IsThumb2) {
11226 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut)
11227 .addReg(Data)
11228 .addReg(AddrIn)
11229 .addImm(StSize)
11231 } else { // arm
11232 BuildMI(*BB, Pos, dl, TII->get(StOpc), AddrOut)
11233 .addReg(Data)
11234 .addReg(AddrIn)
11235 .addReg(0)
11236 .addImm(StSize)
11238 }
11239}
11240
11242ARMTargetLowering::EmitStructByval(MachineInstr &MI,
11243 MachineBasicBlock *BB) const {
11244 // This pseudo instruction has 3 operands: dst, src, size
11245 // We expand it to a loop if size > Subtarget->getMaxInlineSizeThreshold().
11246 // Otherwise, we will generate unrolled scalar copies.
11247 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
11248 const BasicBlock *LLVM_BB = BB->getBasicBlock();
11250
11251 Register dest = MI.getOperand(0).getReg();
11252 Register src = MI.getOperand(1).getReg();
11253 unsigned SizeVal = MI.getOperand(2).getImm();
11254 unsigned Alignment = MI.getOperand(3).getImm();
11255 DebugLoc dl = MI.getDebugLoc();
11256
11257 MachineFunction *MF = BB->getParent();
11258 MachineRegisterInfo &MRI = MF->getRegInfo();
11259 unsigned UnitSize = 0;
11260 const TargetRegisterClass *TRC = nullptr;
11261 const TargetRegisterClass *VecTRC = nullptr;
11262
11263 bool IsThumb1 = Subtarget->isThumb1Only();
11264 bool IsThumb2 = Subtarget->isThumb2();
11265 bool IsThumb = Subtarget->isThumb();
11266
11267 if (Alignment & 1) {
11268 UnitSize = 1;
11269 } else if (Alignment & 2) {
11270 UnitSize = 2;
11271 } else {
11272 // Check whether we can use NEON instructions.
11273 if (!MF->getFunction().hasFnAttribute(Attribute::NoImplicitFloat) &&
11274 Subtarget->hasNEON()) {
11275 if ((Alignment % 16 == 0) && SizeVal >= 16)
11276 UnitSize = 16;
11277 else if ((Alignment % 8 == 0) && SizeVal >= 8)
11278 UnitSize = 8;
11279 }
11280 // Can't use NEON instructions.
11281 if (UnitSize == 0)
11282 UnitSize = 4;
11283 }
11284
11285 // Select the correct opcode and register class for unit size load/store
11286 bool IsNeon = UnitSize >= 8;
11287 TRC = IsThumb ? &ARM::tGPRRegClass : &ARM::GPRRegClass;
11288 if (IsNeon)
11289 VecTRC = UnitSize == 16 ? &ARM::DPairRegClass
11290 : UnitSize == 8 ? &ARM::DPRRegClass
11291 : nullptr;
11292
11293 unsigned BytesLeft = SizeVal % UnitSize;
11294 unsigned LoopSize = SizeVal - BytesLeft;
11295
11296 if (SizeVal <= Subtarget->getMaxInlineSizeThreshold()) {
11297 // Use LDR and STR to copy.
11298 // [scratch, srcOut] = LDR_POST(srcIn, UnitSize)
11299 // [destOut] = STR_POST(scratch, destIn, UnitSize)
11300 unsigned srcIn = src;
11301 unsigned destIn = dest;
11302 for (unsigned i = 0; i < LoopSize; i+=UnitSize) {
11303 Register srcOut = MRI.createVirtualRegister(TRC);
11304 Register destOut = MRI.createVirtualRegister(TRC);
11305 Register scratch = MRI.createVirtualRegister(IsNeon ? VecTRC : TRC);
11306 emitPostLd(BB, MI, TII, dl, UnitSize, scratch, srcIn, srcOut,
11307 IsThumb1, IsThumb2);
11308 emitPostSt(BB, MI, TII, dl, UnitSize, scratch, destIn, destOut,
11309 IsThumb1, IsThumb2);
11310 srcIn = srcOut;
11311 destIn = destOut;
11312 }
11313
11314 // Handle the leftover bytes with LDRB and STRB.
11315 // [scratch, srcOut] = LDRB_POST(srcIn, 1)
11316 // [destOut] = STRB_POST(scratch, destIn, 1)
11317 for (unsigned i = 0; i < BytesLeft; i++) {
11318 Register srcOut = MRI.createVirtualRegister(TRC);
11319 Register destOut = MRI.createVirtualRegister(TRC);
11320 Register scratch = MRI.createVirtualRegister(TRC);
11321 emitPostLd(BB, MI, TII, dl, 1, scratch, srcIn, srcOut,
11322 IsThumb1, IsThumb2);
11323 emitPostSt(BB, MI, TII, dl, 1, scratch, destIn, destOut,
11324 IsThumb1, IsThumb2);
11325 srcIn = srcOut;
11326 destIn = destOut;
11327 }
11328 MI.eraseFromParent(); // The instruction is gone now.
11329 return BB;
11330 }
11331
11332 // Expand the pseudo op to a loop.
11333 // thisMBB:
11334 // ...
11335 // movw varEnd, # --> with thumb2
11336 // movt varEnd, #
11337 // ldrcp varEnd, idx --> without thumb2
11338 // fallthrough --> loopMBB
11339 // loopMBB:
11340 // PHI varPhi, varEnd, varLoop
11341 // PHI srcPhi, src, srcLoop
11342 // PHI destPhi, dst, destLoop
11343 // [scratch, srcLoop] = LDR_POST(srcPhi, UnitSize)
11344 // [destLoop] = STR_POST(scratch, destPhi, UnitSize)
11345 // subs varLoop, varPhi, #UnitSize
11346 // bne loopMBB
11347 // fallthrough --> exitMBB
11348 // exitMBB:
11349 // epilogue to handle left-over bytes
11350 // [scratch, srcOut] = LDRB_POST(srcLoop, 1)
11351 // [destOut] = STRB_POST(scratch, destLoop, 1)
11352 MachineBasicBlock *loopMBB = MF->CreateMachineBasicBlock(LLVM_BB);
11353 MachineBasicBlock *exitMBB = MF->CreateMachineBasicBlock(LLVM_BB);
11354 MF->insert(It, loopMBB);
11355 MF->insert(It, exitMBB);
11356
11357 // Set the call frame size on entry to the new basic blocks.
11358 unsigned CallFrameSize = TII->getCallFrameSizeAt(MI);
11359 loopMBB->setCallFrameSize(CallFrameSize);
11360 exitMBB->setCallFrameSize(CallFrameSize);
11361
11362 // Transfer the remainder of BB and its successor edges to exitMBB.
11363 exitMBB->splice(exitMBB->begin(), BB,
11364 std::next(MachineBasicBlock::iterator(MI)), BB->end());
11366
11367 // Load an immediate to varEnd.
11368 Register varEnd = MRI.createVirtualRegister(TRC);
11369 if (Subtarget->useMovt()) {
11370 BuildMI(BB, dl, TII->get(IsThumb ? ARM::t2MOVi32imm : ARM::MOVi32imm),
11371 varEnd)
11372 .addImm(LoopSize);
11373 } else if (Subtarget->genExecuteOnly()) {
11374 assert(IsThumb && "Non-thumb expected to have used movt");
11375 BuildMI(BB, dl, TII->get(ARM::tMOVi32imm), varEnd).addImm(LoopSize);
11376 } else {
11377 MachineConstantPool *ConstantPool = MF->getConstantPool();
11378 Type *Int32Ty = Type::getInt32Ty(MF->getFunction().getContext());
11379 const Constant *C = ConstantInt::get(Int32Ty, LoopSize);
11380
11381 // MachineConstantPool wants an explicit alignment.
11383 unsigned Idx = ConstantPool->getConstantPoolIndex(C, Alignment);
11384 MachineMemOperand *CPMMO =
11387
11388 if (IsThumb)
11389 BuildMI(*BB, MI, dl, TII->get(ARM::tLDRpci))
11390 .addReg(varEnd, RegState::Define)
11393 .addMemOperand(CPMMO);
11394 else
11395 BuildMI(*BB, MI, dl, TII->get(ARM::LDRcp))
11396 .addReg(varEnd, RegState::Define)
11398 .addImm(0)
11400 .addMemOperand(CPMMO);
11401 }
11402 BB->addSuccessor(loopMBB);
11403
11404 // Generate the loop body:
11405 // varPhi = PHI(varLoop, varEnd)
11406 // srcPhi = PHI(srcLoop, src)
11407 // destPhi = PHI(destLoop, dst)
11408 MachineBasicBlock *entryBB = BB;
11409 BB = loopMBB;
11410 Register varLoop = MRI.createVirtualRegister(TRC);
11411 Register varPhi = MRI.createVirtualRegister(TRC);
11412 Register srcLoop = MRI.createVirtualRegister(TRC);
11413 Register srcPhi = MRI.createVirtualRegister(TRC);
11414 Register destLoop = MRI.createVirtualRegister(TRC);
11415 Register destPhi = MRI.createVirtualRegister(TRC);
11416
11417 BuildMI(*BB, BB->begin(), dl, TII->get(ARM::PHI), varPhi)
11418 .addReg(varLoop).addMBB(loopMBB)
11419 .addReg(varEnd).addMBB(entryBB);
11420 BuildMI(BB, dl, TII->get(ARM::PHI), srcPhi)
11421 .addReg(srcLoop).addMBB(loopMBB)
11422 .addReg(src).addMBB(entryBB);
11423 BuildMI(BB, dl, TII->get(ARM::PHI), destPhi)
11424 .addReg(destLoop).addMBB(loopMBB)
11425 .addReg(dest).addMBB(entryBB);
11426
11427 // [scratch, srcLoop] = LDR_POST(srcPhi, UnitSize)
11428 // [destLoop] = STR_POST(scratch, destPhi, UnitSiz)
11429 Register scratch = MRI.createVirtualRegister(IsNeon ? VecTRC : TRC);
11430 emitPostLd(BB, BB->end(), TII, dl, UnitSize, scratch, srcPhi, srcLoop,
11431 IsThumb1, IsThumb2);
11432 emitPostSt(BB, BB->end(), TII, dl, UnitSize, scratch, destPhi, destLoop,
11433 IsThumb1, IsThumb2);
11434
11435 // Decrement loop variable by UnitSize.
11436 if (IsThumb1) {
11437 BuildMI(*BB, BB->end(), dl, TII->get(ARM::tSUBi8), varLoop)
11438 .add(t1CondCodeOp())
11439 .addReg(varPhi)
11440 .addImm(UnitSize)
11442 } else {
11443 MachineInstrBuilder MIB =
11444 BuildMI(*BB, BB->end(), dl,
11445 TII->get(IsThumb2 ? ARM::t2SUBri : ARM::SUBri), varLoop);
11446 MIB.addReg(varPhi)
11447 .addImm(UnitSize)
11449 .add(condCodeOp());
11450 MIB->getOperand(5).setReg(ARM::CPSR);
11451 MIB->getOperand(5).setIsDef(true);
11452 }
11453 BuildMI(*BB, BB->end(), dl,
11454 TII->get(IsThumb1 ? ARM::tBcc : IsThumb2 ? ARM::t2Bcc : ARM::Bcc))
11455 .addMBB(loopMBB).addImm(ARMCC::NE).addReg(ARM::CPSR);
11456
11457 // loopMBB can loop back to loopMBB or fall through to exitMBB.
11458 BB->addSuccessor(loopMBB);
11459 BB->addSuccessor(exitMBB);
11460
11461 // Add epilogue to handle BytesLeft.
11462 BB = exitMBB;
11463 auto StartOfExit = exitMBB->begin();
11464
11465 // [scratch, srcOut] = LDRB_POST(srcLoop, 1)
11466 // [destOut] = STRB_POST(scratch, destLoop, 1)
11467 unsigned srcIn = srcLoop;
11468 unsigned destIn = destLoop;
11469 for (unsigned i = 0; i < BytesLeft; i++) {
11470 Register srcOut = MRI.createVirtualRegister(TRC);
11471 Register destOut = MRI.createVirtualRegister(TRC);
11472 Register scratch = MRI.createVirtualRegister(TRC);
11473 emitPostLd(BB, StartOfExit, TII, dl, 1, scratch, srcIn, srcOut,
11474 IsThumb1, IsThumb2);
11475 emitPostSt(BB, StartOfExit, TII, dl, 1, scratch, destIn, destOut,
11476 IsThumb1, IsThumb2);
11477 srcIn = srcOut;
11478 destIn = destOut;
11479 }
11480
11481 MI.eraseFromParent(); // The instruction is gone now.
11482 return BB;
11483}
11484
11486ARMTargetLowering::EmitLowered__chkstk(MachineInstr &MI,
11487 MachineBasicBlock *MBB) const {
11488 const TargetMachine &TM = getTargetMachine();
11489 const TargetInstrInfo &TII = *Subtarget->getInstrInfo();
11490 DebugLoc DL = MI.getDebugLoc();
11491
11492 assert(TM.getTargetTriple().isOSWindows() &&
11493 "__chkstk is only supported on Windows");
11494 assert(Subtarget->isThumb2() && "Windows on ARM requires Thumb-2 mode");
11495
11496 // __chkstk takes the number of words to allocate on the stack in R4, and
11497 // returns the stack adjustment in number of bytes in R4. This will not
11498 // clober any other registers (other than the obvious lr).
11499 //
11500 // Although, technically, IP should be considered a register which may be
11501 // clobbered, the call itself will not touch it. Windows on ARM is a pure
11502 // thumb-2 environment, so there is no interworking required. As a result, we
11503 // do not expect a veneer to be emitted by the linker, clobbering IP.
11504 //
11505 // Each module receives its own copy of __chkstk, so no import thunk is
11506 // required, again, ensuring that IP is not clobbered.
11507 //
11508 // Finally, although some linkers may theoretically provide a trampoline for
11509 // out of range calls (which is quite common due to a 32M range limitation of
11510 // branches for Thumb), we can generate the long-call version via
11511 // -mcmodel=large, alleviating the need for the trampoline which may clobber
11512 // IP.
11513
11514 RTLIB::LibcallImpl ChkStkLibcall = getLibcallImpl(RTLIB::STACK_PROBE);
11515 if (ChkStkLibcall == RTLIB::Unsupported)
11516 reportFatalUsageError("no available implementation of __chkstk");
11517
11518 const char *ChkStk = getLibcallImplName(ChkStkLibcall).data();
11519 switch (TM.getCodeModel()) {
11520 case CodeModel::Tiny:
11521 llvm_unreachable("Tiny code model not available on ARM.");
11522 case CodeModel::Small:
11523 case CodeModel::Medium:
11524 case CodeModel::Kernel:
11525 BuildMI(*MBB, MI, DL, TII.get(ARM::tBL))
11527 .addExternalSymbol(ChkStk)
11528 .setOperandDead(3) // implicit-def $lr
11531 .addReg(ARM::R12,
11533 .addReg(ARM::CPSR,
11535 break;
11536 case CodeModel::Large: {
11537 MachineRegisterInfo &MRI = MBB->getParent()->getRegInfo();
11538 Register Reg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11539
11540 BuildMI(*MBB, MI, DL, TII.get(ARM::t2MOVi32imm), Reg)
11541 .addExternalSymbol(ChkStk);
11544 .addReg(Reg)
11545 .setOperandDead(3) // implicit-def $lr
11548 .addReg(ARM::R12,
11550 .addReg(ARM::CPSR,
11552 break;
11553 }
11554 }
11555
11556 BuildMI(*MBB, MI, DL, TII.get(ARM::t2SUBrr), ARM::SP)
11557 .addReg(ARM::SP, RegState::Kill)
11558 .addReg(ARM::R4, RegState::Kill)
11561 .add(condCodeOp());
11562
11563 MI.eraseFromParent();
11564 return MBB;
11565}
11566
11568ARMTargetLowering::EmitLowered__dbzchk(MachineInstr &MI,
11569 MachineBasicBlock *MBB) const {
11570 DebugLoc DL = MI.getDebugLoc();
11571 MachineFunction *MF = MBB->getParent();
11572 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
11573
11574 MachineBasicBlock *ContBB = MF->CreateMachineBasicBlock();
11575 MF->insert(++MBB->getIterator(), ContBB);
11576 ContBB->splice(ContBB->begin(), MBB,
11577 std::next(MachineBasicBlock::iterator(MI)), MBB->end());
11579 MBB->addSuccessor(ContBB);
11580
11581 MachineBasicBlock *TrapBB = MF->CreateMachineBasicBlock();
11582 BuildMI(TrapBB, DL, TII->get(ARM::t__brkdiv0));
11583 MF->push_back(TrapBB);
11584 MBB->addSuccessor(TrapBB);
11585
11586 BuildMI(*MBB, MI, DL, TII->get(ARM::tCMPi8))
11587 .addReg(MI.getOperand(0).getReg())
11588 .addImm(0)
11590 BuildMI(*MBB, MI, DL, TII->get(ARM::t2Bcc))
11591 .addMBB(TrapBB)
11593 .addReg(ARM::CPSR);
11594
11595 MI.eraseFromParent();
11596 return ContBB;
11597}
11598
11599// The CPSR operand of SelectItr might be missing a kill marker
11600// because there were multiple uses of CPSR, and ISel didn't know
11601// which to mark. Figure out whether SelectItr should have had a
11602// kill marker, and set it if it should. Returns the correct kill
11603// marker value.
11606 const TargetRegisterInfo* TRI) {
11607 // Scan forward through BB for a use/def of CPSR.
11608 MachineBasicBlock::iterator miI(std::next(SelectItr));
11609 for (MachineBasicBlock::iterator miE = BB->end(); miI != miE; ++miI) {
11610 const MachineInstr& mi = *miI;
11611 if (mi.readsRegister(ARM::CPSR, /*TRI=*/nullptr))
11612 return false;
11613 if (mi.definesRegister(ARM::CPSR, /*TRI=*/nullptr))
11614 break; // Should have kill-flag - update below.
11615 }
11616
11617 // If we hit the end of the block, check whether CPSR is live into a
11618 // successor.
11619 if (miI == BB->end()) {
11620 for (MachineBasicBlock *Succ : BB->successors())
11621 if (Succ->isLiveIn(ARM::CPSR))
11622 return false;
11623 }
11624
11625 // We found a def, or hit the end of the basic block and CPSR wasn't live
11626 // out. SelectMI should have a kill flag on CPSR.
11627 SelectItr->addRegisterKilled(ARM::CPSR, TRI);
11628 return true;
11629}
11630
11631/// Adds logic in loop entry MBB to calculate loop iteration count and adds
11632/// t2WhileLoopSetup and t2WhileLoopStart to generate WLS loop
11634 MachineBasicBlock *TpLoopBody,
11635 MachineBasicBlock *TpExit, Register OpSizeReg,
11636 const TargetInstrInfo *TII, DebugLoc Dl,
11637 MachineRegisterInfo &MRI) {
11638 // Calculates loop iteration count = ceil(n/16) = (n + 15) >> 4.
11639 Register AddDestReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11640 BuildMI(TpEntry, Dl, TII->get(ARM::t2ADDri), AddDestReg)
11641 .addUse(OpSizeReg)
11642 .addImm(15)
11644 .addReg(0);
11645
11646 Register LsrDestReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11647 BuildMI(TpEntry, Dl, TII->get(ARM::t2LSRri), LsrDestReg)
11648 .addUse(AddDestReg)
11649 .addImm(4)
11651 .addReg(0);
11652
11653 Register TotalIterationsReg = MRI.createVirtualRegister(&ARM::GPRlrRegClass);
11654 BuildMI(TpEntry, Dl, TII->get(ARM::t2WhileLoopSetup), TotalIterationsReg)
11655 .addUse(LsrDestReg);
11656
11657 BuildMI(TpEntry, Dl, TII->get(ARM::t2WhileLoopStart))
11658 .addUse(TotalIterationsReg)
11659 .addMBB(TpExit);
11660
11661 BuildMI(TpEntry, Dl, TII->get(ARM::t2B))
11662 .addMBB(TpLoopBody)
11664
11665 return TotalIterationsReg;
11666}
11667
11668/// Adds logic in the loopBody MBB to generate MVE_VCTP, t2DoLoopDec and
11669/// t2DoLoopEnd. These are used by later passes to generate tail predicated
11670/// loops.
11671static void genTPLoopBody(MachineBasicBlock *TpLoopBody,
11672 MachineBasicBlock *TpEntry, MachineBasicBlock *TpExit,
11673 const TargetInstrInfo *TII, DebugLoc Dl,
11674 MachineRegisterInfo &MRI, Register OpSrcReg,
11675 Register OpDestReg, Register ElementCountReg,
11676 Register TotalIterationsReg, bool IsMemcpy) {
11677 // First insert 4 PHI nodes for: Current pointer to Src (if memcpy), Dest
11678 // array, loop iteration counter, predication counter.
11679
11680 Register SrcPhiReg, CurrSrcReg;
11681 if (IsMemcpy) {
11682 // Current position in the src array
11683 SrcPhiReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11684 CurrSrcReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11685 BuildMI(TpLoopBody, Dl, TII->get(ARM::PHI), SrcPhiReg)
11686 .addUse(OpSrcReg)
11687 .addMBB(TpEntry)
11688 .addUse(CurrSrcReg)
11689 .addMBB(TpLoopBody);
11690 }
11691
11692 // Current position in the dest array
11693 Register DestPhiReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11694 Register CurrDestReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11695 BuildMI(TpLoopBody, Dl, TII->get(ARM::PHI), DestPhiReg)
11696 .addUse(OpDestReg)
11697 .addMBB(TpEntry)
11698 .addUse(CurrDestReg)
11699 .addMBB(TpLoopBody);
11700
11701 // Current loop counter
11702 Register LoopCounterPhiReg = MRI.createVirtualRegister(&ARM::GPRlrRegClass);
11703 Register RemainingLoopIterationsReg =
11704 MRI.createVirtualRegister(&ARM::GPRlrRegClass);
11705 BuildMI(TpLoopBody, Dl, TII->get(ARM::PHI), LoopCounterPhiReg)
11706 .addUse(TotalIterationsReg)
11707 .addMBB(TpEntry)
11708 .addUse(RemainingLoopIterationsReg)
11709 .addMBB(TpLoopBody);
11710
11711 // Predication counter
11712 Register PredCounterPhiReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11713 Register RemainingElementsReg = MRI.createVirtualRegister(&ARM::rGPRRegClass);
11714 BuildMI(TpLoopBody, Dl, TII->get(ARM::PHI), PredCounterPhiReg)
11715 .addUse(ElementCountReg)
11716 .addMBB(TpEntry)
11717 .addUse(RemainingElementsReg)
11718 .addMBB(TpLoopBody);
11719
11720 // Pass predication counter to VCTP
11721 Register VccrReg = MRI.createVirtualRegister(&ARM::VCCRRegClass);
11722 BuildMI(TpLoopBody, Dl, TII->get(ARM::MVE_VCTP8), VccrReg)
11723 .addUse(PredCounterPhiReg)
11725 .addReg(0)
11726 .addReg(0);
11727
11728 BuildMI(TpLoopBody, Dl, TII->get(ARM::t2SUBri), RemainingElementsReg)
11729 .addUse(PredCounterPhiReg)
11730 .addImm(16)
11732 .addReg(0);
11733
11734 // VLDRB (only if memcpy) and VSTRB instructions, predicated using VPR
11735 Register SrcValueReg;
11736 if (IsMemcpy) {
11737 SrcValueReg = MRI.createVirtualRegister(&ARM::MQPRRegClass);
11738 BuildMI(TpLoopBody, Dl, TII->get(ARM::MVE_VLDRBU8_post))
11739 .addDef(CurrSrcReg)
11740 .addDef(SrcValueReg)
11741 .addReg(SrcPhiReg)
11742 .addImm(16)
11744 .addUse(VccrReg)
11745 .addReg(0);
11746 } else
11747 SrcValueReg = OpSrcReg;
11748
11749 BuildMI(TpLoopBody, Dl, TII->get(ARM::MVE_VSTRBU8_post))
11750 .addDef(CurrDestReg)
11751 .addUse(SrcValueReg)
11752 .addReg(DestPhiReg)
11753 .addImm(16)
11755 .addUse(VccrReg)
11756 .addReg(0);
11757
11758 // Add the pseudoInstrs for decrementing the loop counter and marking the
11759 // end:t2DoLoopDec and t2DoLoopEnd
11760 BuildMI(TpLoopBody, Dl, TII->get(ARM::t2LoopDec), RemainingLoopIterationsReg)
11761 .addUse(LoopCounterPhiReg)
11762 .addImm(1);
11763
11764 BuildMI(TpLoopBody, Dl, TII->get(ARM::t2LoopEnd))
11765 .addUse(RemainingLoopIterationsReg)
11766 .addMBB(TpLoopBody);
11767
11768 BuildMI(TpLoopBody, Dl, TII->get(ARM::t2B))
11769 .addMBB(TpExit)
11771}
11772
11774 // KCFI is supported in all ARM/Thumb modes
11775 return true;
11776}
11777
11781 const TargetInstrInfo *TII) const {
11782 assert(MBBI->isCall() && MBBI->getCFIType() &&
11783 "Invalid call instruction for a KCFI check");
11784
11785 MachineOperand *TargetOp = nullptr;
11786 switch (MBBI->getOpcode()) {
11787 // ARM mode opcodes
11788 case ARM::BLX:
11789 case ARM::BLX_pred:
11790 case ARM::BLX_noip:
11791 case ARM::BLX_pred_noip:
11792 case ARM::BX_CALL:
11793 TargetOp = &MBBI->getOperand(0);
11794 break;
11795 case ARM::TCRETURNri:
11796 case ARM::TCRETURNrinotr12:
11797 case ARM::TAILJMPr:
11798 case ARM::TAILJMPr4:
11799 TargetOp = &MBBI->getOperand(0);
11800 break;
11801 // Thumb mode opcodes (Thumb1 and Thumb2)
11802 // Note: Most Thumb call instructions have predicate operands before the
11803 // target register Format: tBLXr pred, predreg, target_register, ...
11804 case ARM::tBLXr: // Thumb1/Thumb2: BLX register (requires V5T)
11805 case ARM::tBLXr_noip: // Thumb1/Thumb2: BLX register, no IP clobber
11806 case ARM::tBX_CALL: // Thumb1 only: BX call (push LR, BX)
11807 TargetOp = &MBBI->getOperand(2);
11808 break;
11809 // Tail call instructions don't have predicates, target is operand 0
11810 case ARM::tTAILJMPr: // Thumb1/Thumb2: Tail call via register
11811 TargetOp = &MBBI->getOperand(0);
11812 break;
11813 default:
11814 llvm_unreachable("Unexpected CFI call opcode");
11815 }
11816
11817 assert(TargetOp && TargetOp->isReg() && "Invalid target operand");
11818 TargetOp->setIsRenamable(false);
11819
11820 // Select the appropriate KCFI_CHECK variant based on the instruction set
11821 unsigned KCFICheckOpcode;
11822 if (Subtarget->isThumb()) {
11823 if (Subtarget->isThumb2()) {
11824 KCFICheckOpcode = ARM::KCFI_CHECK_Thumb2;
11825 } else {
11826 KCFICheckOpcode = ARM::KCFI_CHECK_Thumb1;
11827 }
11828 } else {
11829 KCFICheckOpcode = ARM::KCFI_CHECK_ARM;
11830 }
11831
11832 return BuildMI(MBB, MBBI, MBBI->getDebugLoc(), TII->get(KCFICheckOpcode))
11833 .addReg(TargetOp->getReg())
11834 .addImm(MBBI->getCFIType())
11835 .getInstr();
11836}
11837
11840 MachineBasicBlock *BB) const {
11841 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
11842 DebugLoc dl = MI.getDebugLoc();
11843 bool isThumb2 = Subtarget->isThumb2();
11844 switch (MI.getOpcode()) {
11845 default: {
11846 MI.print(errs());
11847 llvm_unreachable("Unexpected instr type to insert");
11848 }
11849
11850 // Thumb1 post-indexed loads are really just single-register LDMs.
11851 case ARM::tLDR_postidx: {
11852 MachineOperand Def(MI.getOperand(1));
11853 BuildMI(*BB, MI, dl, TII->get(ARM::tLDMIA_UPD))
11854 .add(Def) // Rn_wb
11855 .add(MI.getOperand(2)) // Rn
11856 .add(MI.getOperand(3)) // PredImm
11857 .add(MI.getOperand(4)) // PredReg
11858 .add(MI.getOperand(0)) // Rt
11859 .cloneMemRefs(MI);
11860 MI.eraseFromParent();
11861 return BB;
11862 }
11863
11864 case ARM::MVE_MEMCPYLOOPINST:
11865 case ARM::MVE_MEMSETLOOPINST: {
11866
11867 // Transformation below expands MVE_MEMCPYLOOPINST/MVE_MEMSETLOOPINST Pseudo
11868 // into a Tail Predicated (TP) Loop. It adds the instructions to calculate
11869 // the iteration count =ceil(size_in_bytes/16)) in the TP entry block and
11870 // adds the relevant instructions in the TP loop Body for generation of a
11871 // WLSTP loop.
11872
11873 // Below is relevant portion of the CFG after the transformation.
11874 // The Machine Basic Blocks are shown along with branch conditions (in
11875 // brackets). Note that TP entry/exit MBBs depict the entry/exit of this
11876 // portion of the CFG and may not necessarily be the entry/exit of the
11877 // function.
11878
11879 // (Relevant) CFG after transformation:
11880 // TP entry MBB
11881 // |
11882 // |-----------------|
11883 // (n <= 0) (n > 0)
11884 // | |
11885 // | TP loop Body MBB<--|
11886 // | | |
11887 // \ |___________|
11888 // \ /
11889 // TP exit MBB
11890
11891 MachineFunction *MF = BB->getParent();
11892 MachineFunctionProperties &Properties = MF->getProperties();
11893 MachineRegisterInfo &MRI = MF->getRegInfo();
11894
11895 Register OpDestReg = MI.getOperand(0).getReg();
11896 Register OpSrcReg = MI.getOperand(1).getReg();
11897 Register OpSizeReg = MI.getOperand(2).getReg();
11898
11899 // Allocate the required MBBs and add to parent function.
11900 MachineBasicBlock *TpEntry = BB;
11901 MachineBasicBlock *TpLoopBody = MF->CreateMachineBasicBlock();
11902 MachineBasicBlock *TpExit;
11903
11904 MF->push_back(TpLoopBody);
11905
11906 // If any instructions are present in the current block after
11907 // MVE_MEMCPYLOOPINST or MVE_MEMSETLOOPINST, split the current block and
11908 // move the instructions into the newly created exit block. If there are no
11909 // instructions add an explicit branch to the FallThrough block and then
11910 // split.
11911 //
11912 // The split is required for two reasons:
11913 // 1) A terminator(t2WhileLoopStart) will be placed at that site.
11914 // 2) Since a TPLoopBody will be added later, any phis in successive blocks
11915 // need to be updated. splitAt() already handles this.
11916 TpExit = BB->splitAt(MI, false);
11917 if (TpExit == BB) {
11918 assert(BB->canFallThrough() && "Exit Block must be Fallthrough of the "
11919 "block containing memcpy/memset Pseudo");
11920 TpExit = BB->getFallThrough();
11921 BuildMI(BB, dl, TII->get(ARM::t2B))
11922 .addMBB(TpExit)
11924 TpExit = BB->splitAt(MI, false);
11925 }
11926
11927 // Add logic for iteration count
11928 Register TotalIterationsReg =
11929 genTPEntry(TpEntry, TpLoopBody, TpExit, OpSizeReg, TII, dl, MRI);
11930
11931 // Add the vectorized (and predicated) loads/store instructions
11932 bool IsMemcpy = MI.getOpcode() == ARM::MVE_MEMCPYLOOPINST;
11933 genTPLoopBody(TpLoopBody, TpEntry, TpExit, TII, dl, MRI, OpSrcReg,
11934 OpDestReg, OpSizeReg, TotalIterationsReg, IsMemcpy);
11935
11936 // Required to avoid conflict with the MachineVerifier during testing.
11937 Properties.resetNoPHIs();
11938
11939 // Connect the blocks
11940 TpEntry->addSuccessor(TpLoopBody);
11941 TpLoopBody->addSuccessor(TpLoopBody);
11942 TpLoopBody->addSuccessor(TpExit);
11943
11944 // Reorder for a more natural layout
11945 TpLoopBody->moveAfter(TpEntry);
11946 TpExit->moveAfter(TpLoopBody);
11947
11948 // Finally, remove the memcpy Pseudo Instruction
11949 MI.eraseFromParent();
11950
11951 // Return the exit block as it may contain other instructions requiring a
11952 // custom inserter
11953 return TpExit;
11954 }
11955
11956 // The Thumb2 pre-indexed stores have the same MI operands, they just
11957 // define them differently in the .td files from the isel patterns, so
11958 // they need pseudos.
11959 case ARM::t2STR_preidx:
11960 MI.setDesc(TII->get(ARM::t2STR_PRE));
11961 return BB;
11962 case ARM::t2STRB_preidx:
11963 MI.setDesc(TII->get(ARM::t2STRB_PRE));
11964 return BB;
11965 case ARM::t2STRH_preidx:
11966 MI.setDesc(TII->get(ARM::t2STRH_PRE));
11967 return BB;
11968
11969 case ARM::STRi_preidx:
11970 case ARM::STRBi_preidx: {
11971 unsigned NewOpc = MI.getOpcode() == ARM::STRi_preidx ? ARM::STR_PRE_IMM
11972 : ARM::STRB_PRE_IMM;
11973 // Decode the offset.
11974 unsigned Offset = MI.getOperand(4).getImm();
11975 bool isSub = ARM_AM::getAM2Op(Offset) == ARM_AM::sub;
11977 if (isSub)
11978 Offset = -Offset;
11979
11980 MachineMemOperand *MMO = *MI.memoperands_begin();
11981 BuildMI(*BB, MI, dl, TII->get(NewOpc))
11982 .add(MI.getOperand(0)) // Rn_wb
11983 .add(MI.getOperand(1)) // Rt
11984 .add(MI.getOperand(2)) // Rn
11985 .addImm(Offset) // offset (skip GPR==zero_reg)
11986 .add(MI.getOperand(5)) // pred
11987 .add(MI.getOperand(6))
11988 .addMemOperand(MMO);
11989 MI.eraseFromParent();
11990 return BB;
11991 }
11992 case ARM::STRr_preidx:
11993 case ARM::STRBr_preidx:
11994 case ARM::STRH_preidx: {
11995 unsigned NewOpc;
11996 switch (MI.getOpcode()) {
11997 default: llvm_unreachable("unexpected opcode!");
11998 case ARM::STRr_preidx: NewOpc = ARM::STR_PRE_REG; break;
11999 case ARM::STRBr_preidx: NewOpc = ARM::STRB_PRE_REG; break;
12000 case ARM::STRH_preidx: NewOpc = ARM::STRH_PRE; break;
12001 }
12002 MachineInstrBuilder MIB = BuildMI(*BB, MI, dl, TII->get(NewOpc));
12003 for (const MachineOperand &MO : MI.operands())
12004 MIB.add(MO);
12005 MI.eraseFromParent();
12006 return BB;
12007 }
12008
12009 case ARM::tMOVCCr_pseudo: {
12010 // To "insert" a SELECT_CC instruction, we actually have to insert the
12011 // diamond control-flow pattern. The incoming instruction knows the
12012 // destination vreg to set, the condition code register to branch on, the
12013 // true/false values to select between, and a branch opcode to use.
12014 const BasicBlock *LLVM_BB = BB->getBasicBlock();
12016
12017 // thisMBB:
12018 // ...
12019 // TrueVal = ...
12020 // cmpTY ccX, r1, r2
12021 // bCC copy1MBB
12022 // fallthrough --> copy0MBB
12023 MachineBasicBlock *thisMBB = BB;
12024 MachineFunction *F = BB->getParent();
12025 MachineBasicBlock *copy0MBB = F->CreateMachineBasicBlock(LLVM_BB);
12026 MachineBasicBlock *sinkMBB = F->CreateMachineBasicBlock(LLVM_BB);
12027 F->insert(It, copy0MBB);
12028 F->insert(It, sinkMBB);
12029
12030 // Set the call frame size on entry to the new basic blocks.
12031 unsigned CallFrameSize = TII->getCallFrameSizeAt(MI);
12032 copy0MBB->setCallFrameSize(CallFrameSize);
12033 sinkMBB->setCallFrameSize(CallFrameSize);
12034
12035 // Check whether CPSR is live past the tMOVCCr_pseudo.
12036 const TargetRegisterInfo *TRI = Subtarget->getRegisterInfo();
12037 if (!MI.killsRegister(ARM::CPSR, /*TRI=*/nullptr) &&
12038 !checkAndUpdateCPSRKill(MI, thisMBB, TRI)) {
12039 copy0MBB->addLiveIn(ARM::CPSR);
12040 sinkMBB->addLiveIn(ARM::CPSR);
12041 }
12042
12043 // Transfer the remainder of BB and its successor edges to sinkMBB.
12044 sinkMBB->splice(sinkMBB->begin(), BB,
12045 std::next(MachineBasicBlock::iterator(MI)), BB->end());
12047
12048 BB->addSuccessor(copy0MBB);
12049 BB->addSuccessor(sinkMBB);
12050
12051 BuildMI(BB, dl, TII->get(ARM::tBcc))
12052 .addMBB(sinkMBB)
12053 .addImm(MI.getOperand(3).getImm())
12054 .addReg(MI.getOperand(4).getReg());
12055
12056 // copy0MBB:
12057 // %FalseValue = ...
12058 // # fallthrough to sinkMBB
12059 BB = copy0MBB;
12060
12061 // Update machine-CFG edges
12062 BB->addSuccessor(sinkMBB);
12063
12064 // sinkMBB:
12065 // %Result = phi [ %FalseValue, copy0MBB ], [ %TrueValue, thisMBB ]
12066 // ...
12067 BB = sinkMBB;
12068 BuildMI(*BB, BB->begin(), dl, TII->get(ARM::PHI), MI.getOperand(0).getReg())
12069 .addReg(MI.getOperand(1).getReg())
12070 .addMBB(copy0MBB)
12071 .addReg(MI.getOperand(2).getReg())
12072 .addMBB(thisMBB);
12073
12074 MI.eraseFromParent(); // The pseudo instruction is gone now.
12075 return BB;
12076 }
12077
12078 case ARM::BCCi64:
12079 case ARM::BCCZi64: {
12080 // If there is an unconditional branch to the other successor, remove it.
12081 BB->erase(std::next(MachineBasicBlock::iterator(MI)), BB->end());
12082
12083 // Compare both parts that make up the double comparison separately for
12084 // equality.
12085 bool RHSisZero = MI.getOpcode() == ARM::BCCZi64;
12086
12087 Register LHS1 = MI.getOperand(1).getReg();
12088 Register LHS2 = MI.getOperand(2).getReg();
12089 if (RHSisZero) {
12090 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPri : ARM::CMPri))
12091 .addReg(LHS1)
12092 .addImm(0)
12094 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPri : ARM::CMPri))
12095 .addReg(LHS2).addImm(0)
12096 .addImm(ARMCC::EQ).addReg(ARM::CPSR);
12097 } else {
12098 Register RHS1 = MI.getOperand(3).getReg();
12099 Register RHS2 = MI.getOperand(4).getReg();
12100 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPrr : ARM::CMPrr))
12101 .addReg(LHS1)
12102 .addReg(RHS1)
12104 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2CMPrr : ARM::CMPrr))
12105 .addReg(LHS2).addReg(RHS2)
12106 .addImm(ARMCC::EQ).addReg(ARM::CPSR);
12107 }
12108
12109 MachineBasicBlock *destMBB = MI.getOperand(RHSisZero ? 3 : 5).getMBB();
12110 MachineBasicBlock *exitMBB = OtherSucc(BB, destMBB);
12111 if (MI.getOperand(0).getImm() == ARMCC::NE)
12112 std::swap(destMBB, exitMBB);
12113
12114 BuildMI(BB, dl, TII->get(isThumb2 ? ARM::t2Bcc : ARM::Bcc))
12115 .addMBB(destMBB).addImm(ARMCC::EQ).addReg(ARM::CPSR);
12116 if (isThumb2)
12117 BuildMI(BB, dl, TII->get(ARM::t2B))
12118 .addMBB(exitMBB)
12120 else
12121 BuildMI(BB, dl, TII->get(ARM::B)) .addMBB(exitMBB);
12122
12123 MI.eraseFromParent(); // The pseudo instruction is gone now.
12124 return BB;
12125 }
12126
12127 case ARM::Int_eh_sjlj_setjmp:
12128 case ARM::Int_eh_sjlj_setjmp_nofp:
12129 case ARM::tInt_eh_sjlj_setjmp:
12130 case ARM::t2Int_eh_sjlj_setjmp:
12131 case ARM::t2Int_eh_sjlj_setjmp_nofp:
12132 return BB;
12133
12134 case ARM::Int_eh_sjlj_setup_dispatch:
12135 EmitSjLjDispatchBlock(MI, BB);
12136 return BB;
12137 case ARM::COPY_STRUCT_BYVAL_I32:
12138 ++NumLoopByVals;
12139 return EmitStructByval(MI, BB);
12140 case ARM::WIN__CHKSTK:
12141 return EmitLowered__chkstk(MI, BB);
12142 case ARM::WIN__DBZCHK:
12143 return EmitLowered__dbzchk(MI, BB);
12144 }
12145}
12146
12147/// Attaches vregs to MEMCPY that it will use as scratch registers
12148/// when it is expanded into LDM/STM. This is done as a post-isel lowering
12149/// instead of as a custom inserter because we need the use list from the SDNode.
12150static void attachMEMCPYScratchRegs(const ARMSubtarget *Subtarget,
12151 MachineInstr &MI, const SDNode *Node) {
12152 bool isThumb1 = Subtarget->isThumb1Only();
12153
12154 MachineFunction *MF = MI.getParent()->getParent();
12155 MachineRegisterInfo &MRI = MF->getRegInfo();
12156 MachineInstrBuilder MIB(*MF, MI);
12157
12158 // If the new dst/src is unused mark it as dead.
12159 if (!Node->hasAnyUseOfValue(0)) {
12160 MI.getOperand(0).setIsDead(true);
12161 }
12162 if (!Node->hasAnyUseOfValue(1)) {
12163 MI.getOperand(1).setIsDead(true);
12164 }
12165
12166 // The MEMCPY both defines and kills the scratch registers.
12167 for (unsigned I = 0; I != MI.getOperand(4).getImm(); ++I) {
12168 Register TmpReg = MRI.createVirtualRegister(isThumb1 ? &ARM::tGPRRegClass
12169 : &ARM::GPRRegClass);
12171 }
12172}
12173
12175 SDNode *Node) const {
12176 if (MI.getOpcode() == ARM::MEMCPY) {
12177 attachMEMCPYScratchRegs(Subtarget, MI, Node);
12178 return;
12179 }
12180
12181 const MCInstrDesc *MCID = &MI.getDesc();
12182 // Adjust potentially 's' setting instructions after isel, i.e. ADC, SBC, RSB,
12183 // RSC. Coming out of isel, they have an implicit CPSR def, but the optional
12184 // operand is still set to noreg. If needed, set the optional operand's
12185 // register to CPSR, and remove the redundant implicit def.
12186 //
12187 // e.g. ADCS (..., implicit-def CPSR) -> ADC (... opt:def CPSR).
12188
12189 // Rename pseudo opcodes.
12190 unsigned NewOpc = convertAddSubFlagsOpcode(MI.getOpcode());
12191 unsigned ccOutIdx;
12192 if (NewOpc) {
12193 const ARMBaseInstrInfo *TII = Subtarget->getInstrInfo();
12194 MCID = &TII->get(NewOpc);
12195
12196 assert(MCID->getNumOperands() ==
12197 MI.getDesc().getNumOperands() + 5 - MI.getDesc().getSize()
12198 && "converted opcode should be the same except for cc_out"
12199 " (and, on Thumb1, pred)");
12200
12201 MI.setDesc(*MCID);
12202
12203 // Add the optional cc_out operand
12204 MI.addOperand(MachineOperand::CreateReg(0, /*isDef=*/true));
12205
12206 // On Thumb1, move all input operands to the end, then add the predicate
12207 if (Subtarget->isThumb1Only()) {
12208 for (unsigned c = MCID->getNumOperands() - 4; c--;) {
12209 MI.addOperand(MI.getOperand(1));
12210 MI.removeOperand(1);
12211 }
12212
12213 // Restore the ties
12214 for (unsigned i = MI.getNumOperands(); i--;) {
12215 const MachineOperand& op = MI.getOperand(i);
12216 if (op.isReg() && op.isUse()) {
12217 int DefIdx = MCID->getOperandConstraint(i, MCOI::TIED_TO);
12218 if (DefIdx != -1)
12219 MI.tieOperands(DefIdx, i);
12220 }
12221 }
12222
12224 MI.addOperand(MachineOperand::CreateReg(0, /*isDef=*/false));
12225 ccOutIdx = 1;
12226 } else
12227 ccOutIdx = MCID->getNumOperands() - 1;
12228 } else
12229 ccOutIdx = MCID->getNumOperands() - 1;
12230
12231 // Any ARM instruction that sets the 's' bit should specify an optional
12232 // "cc_out" operand in the last operand position.
12233 if (!MI.hasOptionalDef() || !MCID->operands()[ccOutIdx].isOptionalDef()) {
12234 assert(!NewOpc && "Optional cc_out operand required");
12235 return;
12236 }
12237 // Look for an implicit def of CPSR added by MachineInstr ctor. Remove it
12238 // since we already have an optional CPSR def.
12239 bool definesCPSR = false;
12240 bool deadCPSR = false;
12241 for (unsigned i = MCID->getNumOperands(), e = MI.getNumOperands(); i != e;
12242 ++i) {
12243 const MachineOperand &MO = MI.getOperand(i);
12244 if (MO.isReg() && MO.isDef() && MO.getReg() == ARM::CPSR) {
12245 definesCPSR = true;
12246 if (MO.isDead())
12247 deadCPSR = true;
12248 MI.removeOperand(i);
12249 break;
12250 }
12251 }
12252 if (!definesCPSR) {
12253 assert(!NewOpc && "Optional cc_out operand required");
12254 return;
12255 }
12256 assert(deadCPSR == !Node->hasAnyUseOfValue(1) && "inconsistent dead flag");
12257 if (deadCPSR) {
12258 assert(!MI.getOperand(ccOutIdx).getReg() &&
12259 "expect uninitialized optional cc_out operand");
12260 // Thumb1 instructions must have the S bit even if the CPSR is dead.
12261 if (!Subtarget->isThumb1Only())
12262 return;
12263 }
12264
12265 // If this instruction was defined with an optional CPSR def and its dag node
12266 // had a live implicit CPSR def, then activate the optional CPSR def.
12267 MachineOperand &MO = MI.getOperand(ccOutIdx);
12268 MO.setReg(ARM::CPSR);
12269 MO.setIsDef(true);
12270 MO.setIsDead(deadCPSR);
12271}
12272
12273//===----------------------------------------------------------------------===//
12274// ARM Optimization Hooks
12275//===----------------------------------------------------------------------===//
12276
12277// Helper function that checks if N is a null or all ones constant.
12278static inline bool isZeroOrAllOnes(SDValue N, bool AllOnes) {
12280}
12281
12282// Return true if N is conditionally 0 or all ones.
12283// Detects these expressions where cc is an i1 value:
12284//
12285// (select cc 0, y) [AllOnes=0]
12286// (select cc y, 0) [AllOnes=0]
12287// (zext cc) [AllOnes=0]
12288// (sext cc) [AllOnes=0/1]
12289// (select cc -1, y) [AllOnes=1]
12290// (select cc y, -1) [AllOnes=1]
12291//
12292// Invert is set when N is the null/all ones constant when CC is false.
12293// OtherOp is set to the alternative value of N.
12295 SDValue &CC, bool &Invert,
12296 SDValue &OtherOp,
12297 SelectionDAG &DAG) {
12298 switch (N->getOpcode()) {
12299 default: return false;
12300 case ISD::SELECT: {
12301 CC = N->getOperand(0);
12302 SDValue N1 = N->getOperand(1);
12303 SDValue N2 = N->getOperand(2);
12304 if (isZeroOrAllOnes(N1, AllOnes)) {
12305 Invert = false;
12306 OtherOp = N2;
12307 return true;
12308 }
12309 if (isZeroOrAllOnes(N2, AllOnes)) {
12310 Invert = true;
12311 OtherOp = N1;
12312 return true;
12313 }
12314 return false;
12315 }
12316 case ISD::ZERO_EXTEND:
12317 // (zext cc) can never be the all ones value.
12318 if (AllOnes)
12319 return false;
12320 [[fallthrough]];
12321 case ISD::SIGN_EXTEND: {
12322 SDLoc dl(N);
12323 EVT VT = N->getValueType(0);
12324 CC = N->getOperand(0);
12325 if (CC.getValueType() != MVT::i1 || CC.getOpcode() != ISD::SETCC)
12326 return false;
12327 Invert = !AllOnes;
12328 if (AllOnes)
12329 // When looking for an AllOnes constant, N is an sext, and the 'other'
12330 // value is 0.
12331 OtherOp = DAG.getConstant(0, dl, VT);
12332 else if (N->getOpcode() == ISD::ZERO_EXTEND)
12333 // When looking for a 0 constant, N can be zext or sext.
12334 OtherOp = DAG.getConstant(1, dl, VT);
12335 else
12336 OtherOp = DAG.getAllOnesConstant(dl, VT);
12337 return true;
12338 }
12339 }
12340}
12341
12342// Combine a constant select operand into its use:
12343//
12344// (add (select cc, 0, c), x) -> (select cc, x, (add, x, c))
12345// (sub x, (select cc, 0, c)) -> (select cc, x, (sub, x, c))
12346// (and (select cc, -1, c), x) -> (select cc, x, (and, x, c)) [AllOnes=1]
12347// (or (select cc, 0, c), x) -> (select cc, x, (or, x, c))
12348// (xor (select cc, 0, c), x) -> (select cc, x, (xor, x, c))
12349//
12350// The transform is rejected if the select doesn't have a constant operand that
12351// is null, or all ones when AllOnes is set.
12352//
12353// Also recognize sext/zext from i1:
12354//
12355// (add (zext cc), x) -> (select cc (add x, 1), x)
12356// (add (sext cc), x) -> (select cc (add x, -1), x)
12357//
12358// These transformations eventually create predicated instructions.
12359//
12360// @param N The node to transform.
12361// @param Slct The N operand that is a select.
12362// @param OtherOp The other N operand (x above).
12363// @param DCI Context.
12364// @param AllOnes Require the select constant to be all ones instead of null.
12365// @returns The new node, or SDValue() on failure.
12366static
12369 bool AllOnes = false) {
12370 SelectionDAG &DAG = DCI.DAG;
12371 EVT VT = N->getValueType(0);
12372 SDValue NonConstantVal;
12373 SDValue CCOp;
12374 bool SwapSelectOps;
12375 if (!isConditionalZeroOrAllOnes(Slct.getNode(), AllOnes, CCOp, SwapSelectOps,
12376 NonConstantVal, DAG))
12377 return SDValue();
12378
12379 // Slct is now know to be the desired identity constant when CC is true.
12380 SDValue TrueVal = OtherOp;
12381 SDValue FalseVal = DAG.getNode(N->getOpcode(), SDLoc(N), VT,
12382 OtherOp, NonConstantVal);
12383 // Unless SwapSelectOps says CC should be false.
12384 if (SwapSelectOps)
12385 std::swap(TrueVal, FalseVal);
12386
12387 return DAG.getNode(ISD::SELECT, SDLoc(N), VT,
12388 CCOp, TrueVal, FalseVal);
12389}
12390
12391// Attempt combineSelectAndUse on each operand of a commutative operator N.
12392static
12395 SDValue N0 = N->getOperand(0);
12396 SDValue N1 = N->getOperand(1);
12397 if (N0.getNode()->hasOneUse())
12398 if (SDValue Result = combineSelectAndUse(N, N0, N1, DCI, AllOnes))
12399 return Result;
12400 if (N1.getNode()->hasOneUse())
12401 if (SDValue Result = combineSelectAndUse(N, N1, N0, DCI, AllOnes))
12402 return Result;
12403 return SDValue();
12404}
12405
12407 // VUZP shuffle node.
12408 if (N->getOpcode() == ARMISD::VUZP)
12409 return true;
12410
12411 // "VUZP" on i32 is an alias for VTRN.
12412 if (N->getOpcode() == ARMISD::VTRN && N->getValueType(0) == MVT::v2i32)
12413 return true;
12414
12415 return false;
12416}
12417
12420 const ARMSubtarget *Subtarget) {
12421 // Look for ADD(VUZP.0, VUZP.1).
12422 if (!IsVUZPShuffleNode(N0.getNode()) || N0.getNode() != N1.getNode() ||
12423 N0 == N1)
12424 return SDValue();
12425
12426 // Make sure the ADD is a 64-bit add; there is no 128-bit VPADD.
12427 if (!N->getValueType(0).is64BitVector())
12428 return SDValue();
12429
12430 // Generate vpadd.
12431 SelectionDAG &DAG = DCI.DAG;
12432 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
12433 SDLoc dl(N);
12434 SDNode *Unzip = N0.getNode();
12435 EVT VT = N->getValueType(0);
12436
12438 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpadd, dl,
12439 TLI.getPointerTy(DAG.getDataLayout())));
12440 Ops.push_back(Unzip->getOperand(0));
12441 Ops.push_back(Unzip->getOperand(1));
12442
12443 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, Ops);
12444}
12445
12448 const ARMSubtarget *Subtarget) {
12449 // Check for two extended operands.
12450 if (!(N0.getOpcode() == ISD::SIGN_EXTEND &&
12451 N1.getOpcode() == ISD::SIGN_EXTEND) &&
12452 !(N0.getOpcode() == ISD::ZERO_EXTEND &&
12453 N1.getOpcode() == ISD::ZERO_EXTEND))
12454 return SDValue();
12455
12456 SDValue N00 = N0.getOperand(0);
12457 SDValue N10 = N1.getOperand(0);
12458
12459 // Look for ADD(SEXT(VUZP.0), SEXT(VUZP.1))
12460 if (!IsVUZPShuffleNode(N00.getNode()) || N00.getNode() != N10.getNode() ||
12461 N00 == N10)
12462 return SDValue();
12463
12464 // We only recognize Q register paddl here; this can't be reached until
12465 // after type legalization.
12466 if (!N00.getValueType().is64BitVector() ||
12468 return SDValue();
12469
12470 // Generate vpaddl.
12471 SelectionDAG &DAG = DCI.DAG;
12472 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
12473 SDLoc dl(N);
12474 EVT VT = N->getValueType(0);
12475
12477 // Form vpaddl.sN or vpaddl.uN depending on the kind of extension.
12478 unsigned Opcode;
12479 if (N0.getOpcode() == ISD::SIGN_EXTEND)
12480 Opcode = Intrinsic::arm_neon_vpaddls;
12481 else
12482 Opcode = Intrinsic::arm_neon_vpaddlu;
12483 Ops.push_back(DAG.getConstant(Opcode, dl,
12484 TLI.getPointerTy(DAG.getDataLayout())));
12486 unsigned NumElts = VT.getVectorNumElements();
12487 EVT ConcatVT = EVT::getVectorVT(*DAG.getContext(), ElemTy, NumElts * 2);
12488 SDValue Concat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), ConcatVT,
12489 N00.getOperand(0), N00.getOperand(1));
12490 Ops.push_back(Concat);
12491
12492 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, VT, Ops);
12493}
12494
12495// FIXME: This function shouldn't be necessary; if we lower BUILD_VECTOR in
12496// an appropriate manner, we end up with ADD(VUZP(ZEXT(N))), which is
12497// much easier to match.
12498static SDValue
12501 const ARMSubtarget *Subtarget) {
12502 // Only perform optimization if after legalize, and if NEON is available. We
12503 // also expected both operands to be BUILD_VECTORs.
12504 if (DCI.isBeforeLegalize() || !Subtarget->hasNEON()
12505 || N0.getOpcode() != ISD::BUILD_VECTOR
12506 || N1.getOpcode() != ISD::BUILD_VECTOR)
12507 return SDValue();
12508
12509 // Check output type since VPADDL operand elements can only be 8, 16, or 32.
12510 EVT VT = N->getValueType(0);
12511 if (!VT.isInteger() || VT.getVectorElementType() == MVT::i64)
12512 return SDValue();
12513
12514 // Check that the vector operands are of the right form.
12515 // N0 and N1 are BUILD_VECTOR nodes with N number of EXTRACT_VECTOR
12516 // operands, where N is the size of the formed vector.
12517 // Each EXTRACT_VECTOR should have the same input vector and odd or even
12518 // index such that we have a pair wise add pattern.
12519
12520 // Grab the vector that all EXTRACT_VECTOR nodes should be referencing.
12522 return SDValue();
12523 SDValue Vec = N0->getOperand(0)->getOperand(0);
12524 SDNode *V = Vec.getNode();
12525 unsigned nextIndex = 0;
12526
12527 // For each operands to the ADD which are BUILD_VECTORs,
12528 // check to see if each of their operands are an EXTRACT_VECTOR with
12529 // the same vector and appropriate index.
12530 for (unsigned i = 0, e = N0->getNumOperands(); i != e; ++i) {
12533
12534 SDValue ExtVec0 = N0->getOperand(i);
12535 SDValue ExtVec1 = N1->getOperand(i);
12536
12537 // First operand is the vector, verify its the same.
12538 if (V != ExtVec0->getOperand(0).getNode() ||
12539 V != ExtVec1->getOperand(0).getNode())
12540 return SDValue();
12541
12542 // Second is the constant, verify its correct.
12545
12546 // For the constant, we want to see all the even or all the odd.
12547 if (!C0 || !C1 || C0->getZExtValue() != nextIndex
12548 || C1->getZExtValue() != nextIndex+1)
12549 return SDValue();
12550
12551 // Increment index.
12552 nextIndex+=2;
12553 } else
12554 return SDValue();
12555 }
12556
12557 // Don't generate vpaddl+vmovn; we'll match it to vpadd later. Also make sure
12558 // we're using the entire input vector, otherwise there's a size/legality
12559 // mismatch somewhere.
12560 if (nextIndex != Vec.getValueType().getVectorNumElements() ||
12562 return SDValue();
12563
12564 // Create VPADDL node.
12565 SelectionDAG &DAG = DCI.DAG;
12566 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
12567
12568 SDLoc dl(N);
12569
12570 // Build operand list.
12572 Ops.push_back(DAG.getConstant(Intrinsic::arm_neon_vpaddls, dl,
12573 TLI.getPointerTy(DAG.getDataLayout())));
12574
12575 // Input is the vector.
12576 Ops.push_back(Vec);
12577
12578 // Get widened type and narrowed type.
12579 MVT widenType;
12580 unsigned numElem = VT.getVectorNumElements();
12581
12582 EVT inputLaneType = Vec.getValueType().getVectorElementType();
12583 switch (inputLaneType.getSimpleVT().SimpleTy) {
12584 case MVT::i8: widenType = MVT::getVectorVT(MVT::i16, numElem); break;
12585 case MVT::i16: widenType = MVT::getVectorVT(MVT::i32, numElem); break;
12586 case MVT::i32: widenType = MVT::getVectorVT(MVT::i64, numElem); break;
12587 default:
12588 llvm_unreachable("Invalid vector element type for padd optimization.");
12589 }
12590
12591 SDValue tmp = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, widenType, Ops);
12592 unsigned ExtOp = VT.bitsGT(tmp.getValueType()) ? ISD::ANY_EXTEND : ISD::TRUNCATE;
12593 return DAG.getNode(ExtOp, dl, VT, tmp);
12594}
12595
12597 if (V->getOpcode() == ISD::UMUL_LOHI ||
12598 V->getOpcode() == ISD::SMUL_LOHI)
12599 return V;
12600 return SDValue();
12601}
12602
12603static SDValue AddCombineTo64BitSMLAL16(SDNode *AddcNode, SDNode *AddeNode,
12605 const ARMSubtarget *Subtarget) {
12606 if (!Subtarget->hasBaseDSP())
12607 return SDValue();
12608
12609 // SMLALBB, SMLALBT, SMLALTB, SMLALTT multiply two 16-bit values and
12610 // accumulates the product into a 64-bit value. The 16-bit values will
12611 // be sign extended somehow or SRA'd into 32-bit values
12612 // (addc (adde (mul 16bit, 16bit), lo), hi)
12613 SDValue Mul = AddcNode->getOperand(0);
12614 SDValue Lo = AddcNode->getOperand(1);
12615 if (Mul.getOpcode() != ISD::MUL) {
12616 Lo = AddcNode->getOperand(0);
12617 Mul = AddcNode->getOperand(1);
12618 if (Mul.getOpcode() != ISD::MUL)
12619 return SDValue();
12620 }
12621
12622 SDValue SRA = AddeNode->getOperand(0);
12623 SDValue Hi = AddeNode->getOperand(1);
12624 if (SRA.getOpcode() != ISD::SRA) {
12625 SRA = AddeNode->getOperand(1);
12626 Hi = AddeNode->getOperand(0);
12627 if (SRA.getOpcode() != ISD::SRA)
12628 return SDValue();
12629 }
12630 if (auto Const = dyn_cast<ConstantSDNode>(SRA.getOperand(1))) {
12631 if (Const->getZExtValue() != 31)
12632 return SDValue();
12633 } else
12634 return SDValue();
12635
12636 if (SRA.getOperand(0) != Mul)
12637 return SDValue();
12638
12639 SelectionDAG &DAG = DCI.DAG;
12640 SDLoc dl(AddcNode);
12641 unsigned Opcode = 0;
12642 SDValue Op0;
12643 SDValue Op1;
12644
12645 if (isS16(Mul.getOperand(0), DAG) && isS16(Mul.getOperand(1), DAG)) {
12646 Opcode = ARMISD::SMLALBB;
12647 Op0 = Mul.getOperand(0);
12648 Op1 = Mul.getOperand(1);
12649 } else if (isS16(Mul.getOperand(0), DAG) && isSRA16(Mul.getOperand(1))) {
12650 Opcode = ARMISD::SMLALBT;
12651 Op0 = Mul.getOperand(0);
12652 Op1 = Mul.getOperand(1).getOperand(0);
12653 } else if (isSRA16(Mul.getOperand(0)) && isS16(Mul.getOperand(1), DAG)) {
12654 Opcode = ARMISD::SMLALTB;
12655 Op0 = Mul.getOperand(0).getOperand(0);
12656 Op1 = Mul.getOperand(1);
12657 } else if (isSRA16(Mul.getOperand(0)) && isSRA16(Mul.getOperand(1))) {
12658 Opcode = ARMISD::SMLALTT;
12659 Op0 = Mul->getOperand(0).getOperand(0);
12660 Op1 = Mul->getOperand(1).getOperand(0);
12661 }
12662
12663 if (!Op0 || !Op1)
12664 return SDValue();
12665
12666 SDValue SMLAL = DAG.getNode(Opcode, dl, DAG.getVTList(MVT::i32, MVT::i32),
12667 Op0, Op1, Lo, Hi);
12668 // Replace the ADDs' nodes uses by the MLA node's values.
12669 SDValue HiMLALResult(SMLAL.getNode(), 1);
12670 SDValue LoMLALResult(SMLAL.getNode(), 0);
12671
12672 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcNode, 0), LoMLALResult);
12673 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeNode, 0), HiMLALResult);
12674
12675 // Return original node to notify the driver to stop replacing.
12676 SDValue resNode(AddcNode, 0);
12677 return resNode;
12678}
12679
12682 const ARMSubtarget *Subtarget) {
12683 // Look for multiply add opportunities.
12684 // The pattern is a ISD::UMUL_LOHI followed by two add nodes, where
12685 // each add nodes consumes a value from ISD::UMUL_LOHI and there is
12686 // a glue link from the first add to the second add.
12687 // If we find this pattern, we can replace the U/SMUL_LOHI, ADDC, and ADDE by
12688 // a S/UMLAL instruction.
12689 // UMUL_LOHI
12690 // / :lo \ :hi
12691 // V \ [no multiline comment]
12692 // loAdd -> ADDC |
12693 // \ :carry /
12694 // V V
12695 // ADDE <- hiAdd
12696 //
12697 // In the special case where only the higher part of a signed result is used
12698 // and the add to the low part of the result of ISD::UMUL_LOHI adds or subtracts
12699 // a constant with the exact value of 0x80000000, we recognize we are dealing
12700 // with a "rounded multiply and add" (or subtract) and transform it into
12701 // either a ARMISD::SMMLAR or ARMISD::SMMLSR respectively.
12702
12703 assert((AddeSubeNode->getOpcode() == ARMISD::ADDE ||
12704 AddeSubeNode->getOpcode() == ARMISD::SUBE) &&
12705 "Expect an ADDE or SUBE");
12706
12707 assert(AddeSubeNode->getNumOperands() == 3 &&
12708 AddeSubeNode->getOperand(2).getValueType() == MVT::i32 &&
12709 "ADDE node has the wrong inputs");
12710
12711 // Check that we are chained to the right ADDC or SUBC node.
12712 SDNode *AddcSubcNode = AddeSubeNode->getOperand(2).getNode();
12713 if ((AddeSubeNode->getOpcode() == ARMISD::ADDE &&
12714 AddcSubcNode->getOpcode() != ARMISD::ADDC) ||
12715 (AddeSubeNode->getOpcode() == ARMISD::SUBE &&
12716 AddcSubcNode->getOpcode() != ARMISD::SUBC))
12717 return SDValue();
12718
12719 SDValue AddcSubcOp0 = AddcSubcNode->getOperand(0);
12720 SDValue AddcSubcOp1 = AddcSubcNode->getOperand(1);
12721
12722 // Check if the two operands are from the same mul_lohi node.
12723 if (AddcSubcOp0.getNode() == AddcSubcOp1.getNode())
12724 return SDValue();
12725
12726 assert(AddcSubcNode->getNumValues() == 2 &&
12727 AddcSubcNode->getValueType(0) == MVT::i32 &&
12728 "Expect ADDC with two result values. First: i32");
12729
12730 // Check that the ADDC adds the low result of the S/UMUL_LOHI. If not, it
12731 // maybe a SMLAL which multiplies two 16-bit values.
12732 if (AddeSubeNode->getOpcode() == ARMISD::ADDE &&
12733 AddcSubcOp0->getOpcode() != ISD::UMUL_LOHI &&
12734 AddcSubcOp0->getOpcode() != ISD::SMUL_LOHI &&
12735 AddcSubcOp1->getOpcode() != ISD::UMUL_LOHI &&
12736 AddcSubcOp1->getOpcode() != ISD::SMUL_LOHI)
12737 return AddCombineTo64BitSMLAL16(AddcSubcNode, AddeSubeNode, DCI, Subtarget);
12738
12739 // Check for the triangle shape.
12740 SDValue AddeSubeOp0 = AddeSubeNode->getOperand(0);
12741 SDValue AddeSubeOp1 = AddeSubeNode->getOperand(1);
12742
12743 // Make sure that the ADDE/SUBE operands are not coming from the same node.
12744 if (AddeSubeOp0.getNode() == AddeSubeOp1.getNode())
12745 return SDValue();
12746
12747 // Find the MUL_LOHI node walking up ADDE/SUBE's operands.
12748 bool IsLeftOperandMUL = false;
12749 SDValue MULOp = findMUL_LOHI(AddeSubeOp0);
12750 if (MULOp == SDValue())
12751 MULOp = findMUL_LOHI(AddeSubeOp1);
12752 else
12753 IsLeftOperandMUL = true;
12754 if (MULOp == SDValue())
12755 return SDValue();
12756
12757 // Figure out the right opcode.
12758 unsigned Opc = MULOp->getOpcode();
12759 unsigned FinalOpc = (Opc == ISD::SMUL_LOHI) ? ARMISD::SMLAL : ARMISD::UMLAL;
12760
12761 // Figure out the high and low input values to the MLAL node.
12762 SDValue *HiAddSub = nullptr;
12763 SDValue *LoMul = nullptr;
12764 SDValue *LowAddSub = nullptr;
12765
12766 // Ensure that ADDE/SUBE is from high result of ISD::xMUL_LOHI.
12767 if ((AddeSubeOp0 != MULOp.getValue(1)) && (AddeSubeOp1 != MULOp.getValue(1)))
12768 return SDValue();
12769
12770 if (IsLeftOperandMUL)
12771 HiAddSub = &AddeSubeOp1;
12772 else
12773 HiAddSub = &AddeSubeOp0;
12774
12775 // Ensure that LoMul and LowAddSub are taken from correct ISD::SMUL_LOHI node
12776 // whose low result is fed to the ADDC/SUBC we are checking.
12777
12778 if (AddcSubcOp0 == MULOp.getValue(0)) {
12779 LoMul = &AddcSubcOp0;
12780 LowAddSub = &AddcSubcOp1;
12781 }
12782 if (AddcSubcOp1 == MULOp.getValue(0)) {
12783 LoMul = &AddcSubcOp1;
12784 LowAddSub = &AddcSubcOp0;
12785 }
12786
12787 if (!LoMul)
12788 return SDValue();
12789
12790 // If HiAddSub is the same node as ADDC/SUBC or is a predecessor of ADDC/SUBC
12791 // the replacement below will create a cycle.
12792 if (AddcSubcNode == HiAddSub->getNode() ||
12793 AddcSubcNode->isPredecessorOf(HiAddSub->getNode()))
12794 return SDValue();
12795
12796 // Create the merged node.
12797 SelectionDAG &DAG = DCI.DAG;
12798
12799 // Start building operand list.
12801 Ops.push_back(LoMul->getOperand(0));
12802 Ops.push_back(LoMul->getOperand(1));
12803
12804 // Check whether we can use SMMLAR, SMMLSR or SMMULR instead. For this to be
12805 // the case, we must be doing signed multiplication and only use the higher
12806 // part of the result of the MLAL, furthermore the LowAddSub must be a constant
12807 // addition or subtraction with the value of 0x800000.
12808 if (Subtarget->hasV6Ops() && Subtarget->hasDSP() && Subtarget->useMulOps() &&
12809 FinalOpc == ARMISD::SMLAL && !AddeSubeNode->hasAnyUseOfValue(1) &&
12810 LowAddSub->getNode()->getOpcode() == ISD::Constant &&
12811 static_cast<ConstantSDNode *>(LowAddSub->getNode())->getZExtValue() ==
12812 0x80000000) {
12813 Ops.push_back(*HiAddSub);
12814 if (AddcSubcNode->getOpcode() == ARMISD::SUBC) {
12815 FinalOpc = ARMISD::SMMLSR;
12816 } else {
12817 FinalOpc = ARMISD::SMMLAR;
12818 }
12819 SDValue NewNode = DAG.getNode(FinalOpc, SDLoc(AddcSubcNode), MVT::i32, Ops);
12820 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeSubeNode, 0), NewNode);
12821
12822 return SDValue(AddeSubeNode, 0);
12823 } else if (AddcSubcNode->getOpcode() == ARMISD::SUBC)
12824 // SMMLS is generated during instruction selection and the rest of this
12825 // function can not handle the case where AddcSubcNode is a SUBC.
12826 return SDValue();
12827
12828 // Finish building the operand list for {U/S}MLAL
12829 Ops.push_back(*LowAddSub);
12830 Ops.push_back(*HiAddSub);
12831
12832 SDValue MLALNode = DAG.getNode(FinalOpc, SDLoc(AddcSubcNode),
12833 DAG.getVTList(MVT::i32, MVT::i32), Ops);
12834
12835 // Replace the ADDs' nodes uses by the MLA node's values.
12836 SDValue HiMLALResult(MLALNode.getNode(), 1);
12837 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeSubeNode, 0), HiMLALResult);
12838
12839 SDValue LoMLALResult(MLALNode.getNode(), 0);
12840 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcSubcNode, 0), LoMLALResult);
12841
12842 // Return original node to notify the driver to stop replacing.
12843 return SDValue(AddeSubeNode, 0);
12844}
12845
12848 const ARMSubtarget *Subtarget) {
12849 // UMAAL is similar to UMLAL except that it adds two unsigned values.
12850 // While trying to combine for the other MLAL nodes, first search for the
12851 // chance to use UMAAL. Check if Addc uses a node which has already
12852 // been combined into a UMLAL. The other pattern is UMLAL using Addc/Adde
12853 // as the addend, and it's handled in PerformUMLALCombine.
12854
12855 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP())
12856 return AddCombineTo64bitMLAL(AddeNode, DCI, Subtarget);
12857
12858 // Check that we have a glued ADDC node.
12859 SDNode* AddcNode = AddeNode->getOperand(2).getNode();
12860 if (AddcNode->getOpcode() != ARMISD::ADDC)
12861 return SDValue();
12862
12863 // Find the converted UMAAL or quit if it doesn't exist.
12864 SDNode *UmlalNode = nullptr;
12865 SDValue AddHi;
12866 if (AddcNode->getOperand(0).getOpcode() == ARMISD::UMLAL) {
12867 UmlalNode = AddcNode->getOperand(0).getNode();
12868 AddHi = AddcNode->getOperand(1);
12869 } else if (AddcNode->getOperand(1).getOpcode() == ARMISD::UMLAL) {
12870 UmlalNode = AddcNode->getOperand(1).getNode();
12871 AddHi = AddcNode->getOperand(0);
12872 } else {
12873 return AddCombineTo64bitMLAL(AddeNode, DCI, Subtarget);
12874 }
12875
12876 // The ADDC should be glued to an ADDE node, which uses the same UMLAL as
12877 // the ADDC as well as Zero.
12878 if (!isNullConstant(UmlalNode->getOperand(3)))
12879 return SDValue();
12880
12881 if ((isNullConstant(AddeNode->getOperand(0)) &&
12882 AddeNode->getOperand(1).getNode() == UmlalNode) ||
12883 (AddeNode->getOperand(0).getNode() == UmlalNode &&
12884 isNullConstant(AddeNode->getOperand(1)))) {
12885 SelectionDAG &DAG = DCI.DAG;
12886 SDValue Ops[] = { UmlalNode->getOperand(0), UmlalNode->getOperand(1),
12887 UmlalNode->getOperand(2), AddHi };
12888 SDValue UMAAL = DAG.getNode(ARMISD::UMAAL, SDLoc(AddcNode),
12889 DAG.getVTList(MVT::i32, MVT::i32), Ops);
12890
12891 // Replace the ADDs' nodes uses by the UMAAL node's values.
12892 DAG.ReplaceAllUsesOfValueWith(SDValue(AddeNode, 0), SDValue(UMAAL.getNode(), 1));
12893 DAG.ReplaceAllUsesOfValueWith(SDValue(AddcNode, 0), SDValue(UMAAL.getNode(), 0));
12894
12895 // Return original node to notify the driver to stop replacing.
12896 return SDValue(AddeNode, 0);
12897 }
12898 return SDValue();
12899}
12900
12902 const ARMSubtarget *Subtarget) {
12903 if (!Subtarget->hasV6Ops() || !Subtarget->hasDSP())
12904 return SDValue();
12905
12906 // Check that we have a pair of ADDC and ADDE as operands.
12907 // Both addends of the ADDE must be zero.
12908 SDNode* AddcNode = N->getOperand(2).getNode();
12909 SDNode* AddeNode = N->getOperand(3).getNode();
12910 if ((AddcNode->getOpcode() == ARMISD::ADDC) &&
12911 (AddeNode->getOpcode() == ARMISD::ADDE) &&
12912 isNullConstant(AddeNode->getOperand(0)) &&
12913 isNullConstant(AddeNode->getOperand(1)) &&
12914 (AddeNode->getOperand(2).getNode() == AddcNode))
12915 return DAG.getNode(ARMISD::UMAAL, SDLoc(N),
12916 DAG.getVTList(MVT::i32, MVT::i32),
12917 {N->getOperand(0), N->getOperand(1),
12918 AddcNode->getOperand(0), AddcNode->getOperand(1)});
12919 else
12920 return SDValue();
12921}
12922
12925 const ARMSubtarget *Subtarget) {
12926 SelectionDAG &DAG(DCI.DAG);
12927
12928 if (N->getOpcode() == ARMISD::SUBC && N->hasAnyUseOfValue(1)) {
12929 // (SUBC (ADDE 0, 0, C), 1) -> C
12930 SDValue LHS = N->getOperand(0);
12931 SDValue RHS = N->getOperand(1);
12932 if (LHS->getOpcode() == ARMISD::ADDE &&
12933 isNullConstant(LHS->getOperand(0)) &&
12934 isNullConstant(LHS->getOperand(1)) && isOneConstant(RHS)) {
12935 return DCI.CombineTo(N, SDValue(N, 0), LHS->getOperand(2));
12936 }
12937 }
12938
12939 if (Subtarget->isThumb1Only()) {
12940 SDValue RHS = N->getOperand(1);
12942 int32_t imm = C->getSExtValue();
12943 if (imm < 0 && imm > std::numeric_limits<int>::min()) {
12944 SDLoc DL(N);
12945 RHS = DAG.getConstant(-imm, DL, MVT::i32);
12946 unsigned Opcode = (N->getOpcode() == ARMISD::ADDC) ? ARMISD::SUBC
12947 : ARMISD::ADDC;
12948 return DAG.getNode(Opcode, DL, N->getVTList(), N->getOperand(0), RHS);
12949 }
12950 }
12951 }
12952
12953 return SDValue();
12954}
12955
12958 const ARMSubtarget *Subtarget) {
12959 if (Subtarget->isThumb1Only()) {
12960 SelectionDAG &DAG = DCI.DAG;
12961 SDValue RHS = N->getOperand(1);
12963 int64_t imm = C->getSExtValue();
12964 if (imm < 0) {
12965 SDLoc DL(N);
12966
12967 // The with-carry-in form matches bitwise not instead of the negation.
12968 // Effectively, the inverse interpretation of the carry flag already
12969 // accounts for part of the negation.
12970 RHS = DAG.getConstant(~imm, DL, MVT::i32);
12971
12972 unsigned Opcode = (N->getOpcode() == ARMISD::ADDE) ? ARMISD::SUBE
12973 : ARMISD::ADDE;
12974 return DAG.getNode(Opcode, DL, N->getVTList(),
12975 N->getOperand(0), RHS, N->getOperand(2));
12976 }
12977 }
12978 } else if (N->getOperand(1)->getOpcode() == ISD::SMUL_LOHI) {
12979 return AddCombineTo64bitMLAL(N, DCI, Subtarget);
12980 }
12981 return SDValue();
12982}
12983
12986 const ARMSubtarget *Subtarget) {
12987 if (!Subtarget->hasMVEIntegerOps())
12988 return SDValue();
12989
12990 SDLoc dl(N);
12991 SDValue SetCC;
12992 SDValue LHS;
12993 SDValue RHS;
12994 ISD::CondCode CC;
12995 SDValue TrueVal;
12996 SDValue FalseVal;
12997
12998 if (N->getOpcode() == ISD::SELECT &&
12999 N->getOperand(0)->getOpcode() == ISD::SETCC) {
13000 SetCC = N->getOperand(0);
13001 LHS = SetCC->getOperand(0);
13002 RHS = SetCC->getOperand(1);
13003 CC = cast<CondCodeSDNode>(SetCC->getOperand(2))->get();
13004 TrueVal = N->getOperand(1);
13005 FalseVal = N->getOperand(2);
13006 } else if (N->getOpcode() == ISD::SELECT_CC) {
13007 LHS = N->getOperand(0);
13008 RHS = N->getOperand(1);
13009 CC = cast<CondCodeSDNode>(N->getOperand(4))->get();
13010 TrueVal = N->getOperand(2);
13011 FalseVal = N->getOperand(3);
13012 } else {
13013 return SDValue();
13014 }
13015
13016 unsigned int Opcode = 0;
13017 if ((TrueVal->getOpcode() == ISD::VECREDUCE_UMIN ||
13018 FalseVal->getOpcode() == ISD::VECREDUCE_UMIN) &&
13019 (CC == ISD::SETULT || CC == ISD::SETUGT)) {
13020 Opcode = ARMISD::VMINVu;
13021 if (CC == ISD::SETUGT)
13022 std::swap(TrueVal, FalseVal);
13023 } else if ((TrueVal->getOpcode() == ISD::VECREDUCE_SMIN ||
13024 FalseVal->getOpcode() == ISD::VECREDUCE_SMIN) &&
13025 (CC == ISD::SETLT || CC == ISD::SETGT)) {
13026 Opcode = ARMISD::VMINVs;
13027 if (CC == ISD::SETGT)
13028 std::swap(TrueVal, FalseVal);
13029 } else if ((TrueVal->getOpcode() == ISD::VECREDUCE_UMAX ||
13030 FalseVal->getOpcode() == ISD::VECREDUCE_UMAX) &&
13031 (CC == ISD::SETUGT || CC == ISD::SETULT)) {
13032 Opcode = ARMISD::VMAXVu;
13033 if (CC == ISD::SETULT)
13034 std::swap(TrueVal, FalseVal);
13035 } else if ((TrueVal->getOpcode() == ISD::VECREDUCE_SMAX ||
13036 FalseVal->getOpcode() == ISD::VECREDUCE_SMAX) &&
13037 (CC == ISD::SETGT || CC == ISD::SETLT)) {
13038 Opcode = ARMISD::VMAXVs;
13039 if (CC == ISD::SETLT)
13040 std::swap(TrueVal, FalseVal);
13041 } else
13042 return SDValue();
13043
13044 // Normalise to the right hand side being the vector reduction
13045 switch (TrueVal->getOpcode()) {
13050 std::swap(LHS, RHS);
13051 std::swap(TrueVal, FalseVal);
13052 break;
13053 }
13054
13055 EVT VectorType = FalseVal->getOperand(0).getValueType();
13056
13057 if (VectorType != MVT::v16i8 && VectorType != MVT::v8i16 &&
13058 VectorType != MVT::v4i32)
13059 return SDValue();
13060
13061 EVT VectorScalarType = VectorType.getVectorElementType();
13062
13063 // The values being selected must also be the ones being compared
13064 if (TrueVal != LHS || FalseVal != RHS)
13065 return SDValue();
13066
13067 EVT LeftType = LHS->getValueType(0);
13068 EVT RightType = RHS->getValueType(0);
13069
13070 // The types must match the reduced type too
13071 if (LeftType != VectorScalarType || RightType != VectorScalarType)
13072 return SDValue();
13073
13074 // Legalise the scalar to an i32
13075 if (VectorScalarType != MVT::i32)
13076 LHS = DCI.DAG.getNode(ISD::ANY_EXTEND, dl, MVT::i32, LHS);
13077
13078 // Generate the reduction as an i32 for legalisation purposes
13079 auto Reduction =
13080 DCI.DAG.getNode(Opcode, dl, MVT::i32, LHS, RHS->getOperand(0));
13081
13082 // The result isn't actually an i32 so truncate it back to its original type
13083 if (VectorScalarType != MVT::i32)
13084 Reduction = DCI.DAG.getNode(ISD::TRUNCATE, dl, VectorScalarType, Reduction);
13085
13086 return Reduction;
13087}
13088
13089// A special combine for the vqdmulh family of instructions. This is one of the
13090// potential set of patterns that could patch this instruction. The base pattern
13091// you would expect to be min(max(ashr(mul(mul(sext(x), 2), sext(y)), 16))).
13092// This matches the different min(max(ashr(mul(mul(sext(x), sext(y)), 2), 16))),
13093// which llvm will have optimized to min(ashr(mul(sext(x), sext(y)), 15))) as
13094// the max is unnecessary.
13096 EVT VT = N->getValueType(0);
13097 SDValue Shft;
13098 ConstantSDNode *Clamp;
13099
13100 if (!VT.isVector() || VT.getScalarSizeInBits() > 64)
13101 return SDValue();
13102
13103 if (N->getOpcode() == ISD::SMIN) {
13104 Shft = N->getOperand(0);
13105 Clamp = isConstOrConstSplat(N->getOperand(1));
13106 } else if (N->getOpcode() == ISD::VSELECT) {
13107 // Detect a SMIN, which for an i64 node will be a vselect/setcc, not a smin.
13108 SDValue Cmp = N->getOperand(0);
13109 if (Cmp.getOpcode() != ISD::SETCC ||
13110 cast<CondCodeSDNode>(Cmp.getOperand(2))->get() != ISD::SETLT ||
13111 Cmp.getOperand(0) != N->getOperand(1) ||
13112 Cmp.getOperand(1) != N->getOperand(2))
13113 return SDValue();
13114 Shft = N->getOperand(1);
13115 Clamp = isConstOrConstSplat(N->getOperand(2));
13116 } else
13117 return SDValue();
13118
13119 if (!Clamp)
13120 return SDValue();
13121
13122 MVT ScalarType;
13123 int ShftAmt = 0;
13124 switch (Clamp->getSExtValue()) {
13125 case (1 << 7) - 1:
13126 ScalarType = MVT::i8;
13127 ShftAmt = 7;
13128 break;
13129 case (1 << 15) - 1:
13130 ScalarType = MVT::i16;
13131 ShftAmt = 15;
13132 break;
13133 case (1ULL << 31) - 1:
13134 ScalarType = MVT::i32;
13135 ShftAmt = 31;
13136 break;
13137 default:
13138 return SDValue();
13139 }
13140
13141 if (Shft.getOpcode() != ISD::SRA)
13142 return SDValue();
13144 if (!N1 || N1->getSExtValue() != ShftAmt)
13145 return SDValue();
13146
13147 SDValue Mul = Shft.getOperand(0);
13148 if (Mul.getOpcode() != ISD::MUL)
13149 return SDValue();
13150
13151 SDValue Ext0 = Mul.getOperand(0);
13152 SDValue Ext1 = Mul.getOperand(1);
13153 if (Ext0.getOpcode() != ISD::SIGN_EXTEND ||
13154 Ext1.getOpcode() != ISD::SIGN_EXTEND)
13155 return SDValue();
13156 EVT VecVT = Ext0.getOperand(0).getValueType();
13157 if (!VecVT.isPow2VectorType() || VecVT.getVectorNumElements() == 1)
13158 return SDValue();
13159 if (Ext1.getOperand(0).getValueType() != VecVT ||
13160 VecVT.getScalarType() != ScalarType ||
13161 VT.getScalarSizeInBits() < ScalarType.getScalarSizeInBits() * 2)
13162 return SDValue();
13163
13164 SDLoc DL(Mul);
13165 unsigned LegalLanes = 128 / (ShftAmt + 1);
13166 EVT LegalVecVT = MVT::getVectorVT(ScalarType, LegalLanes);
13167 // For types smaller than legal vectors extend to be legal and only use needed
13168 // lanes.
13169 if (VecVT.getSizeInBits() < 128) {
13170 EVT ExtVecVT =
13172 VecVT.getVectorNumElements());
13173 SDValue Inp0 =
13174 DAG.getNode(ISD::ANY_EXTEND, DL, ExtVecVT, Ext0.getOperand(0));
13175 SDValue Inp1 =
13176 DAG.getNode(ISD::ANY_EXTEND, DL, ExtVecVT, Ext1.getOperand(0));
13177 Inp0 = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, LegalVecVT, Inp0);
13178 Inp1 = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, LegalVecVT, Inp1);
13179 SDValue VQDMULH = DAG.getNode(ARMISD::VQDMULH, DL, LegalVecVT, Inp0, Inp1);
13180 SDValue Trunc = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, ExtVecVT, VQDMULH);
13181 Trunc = DAG.getNode(ISD::TRUNCATE, DL, VecVT, Trunc);
13182 return DAG.getNode(ISD::SIGN_EXTEND, DL, VT, Trunc);
13183 }
13184
13185 // For larger types, split into legal sized chunks.
13186 assert(VecVT.getSizeInBits() % 128 == 0 && "Expected a power2 type");
13187 unsigned NumParts = VecVT.getSizeInBits() / 128;
13189 for (unsigned I = 0; I < NumParts; ++I) {
13190 SDValue Inp0 =
13191 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, LegalVecVT, Ext0.getOperand(0),
13192 DAG.getVectorIdxConstant(I * LegalLanes, DL));
13193 SDValue Inp1 =
13194 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, LegalVecVT, Ext1.getOperand(0),
13195 DAG.getVectorIdxConstant(I * LegalLanes, DL));
13196 SDValue VQDMULH = DAG.getNode(ARMISD::VQDMULH, DL, LegalVecVT, Inp0, Inp1);
13197 Parts.push_back(VQDMULH);
13198 }
13199 return DAG.getNode(ISD::SIGN_EXTEND, DL, VT,
13200 DAG.getNode(ISD::CONCAT_VECTORS, DL, VecVT, Parts));
13201}
13202
13205 const ARMSubtarget *Subtarget) {
13206 if (!Subtarget->hasMVEIntegerOps())
13207 return SDValue();
13208
13209 // Constant fold vselect 0, A, B -> B
13210 // and vselect 0xffff, A, B -> A
13211 if (N->getOperand(0).getOpcode() == ARMISD::PREDICATE_CAST &&
13212 isa<ConstantSDNode>(N->getOperand(0).getOperand(0))) {
13213 unsigned C = N->getOperand(0).getConstantOperandVal(0);
13214 if (C == 0)
13215 return N->getOperand(2);
13216 if (C == 0xffff)
13217 return N->getOperand(1);
13218 }
13219
13220 if (SDValue V = PerformVQDMULHCombine(N, DCI.DAG))
13221 return V;
13222
13223 // Transforms vselect(not(cond), lhs, rhs) into vselect(cond, rhs, lhs).
13224 //
13225 // We need to re-implement this optimization here as the implementation in the
13226 // Target-Independent DAGCombiner does not handle the kind of constant we make
13227 // (it calls isConstOrConstSplat with AllowTruncation set to false - and for
13228 // good reason, allowing truncation there would break other targets).
13229 //
13230 // Currently, this is only done for MVE, as it's the only target that benefits
13231 // from this transformation (e.g. VPNOT+VPSEL becomes a single VPSEL).
13232 if (N->getOperand(0).getOpcode() != ISD::XOR)
13233 return SDValue();
13234 SDValue XOR = N->getOperand(0);
13235
13236 // Check if the XOR's RHS is either a 1, or a BUILD_VECTOR of 1s.
13237 // It is important to check with truncation allowed as the BUILD_VECTORs we
13238 // generate in those situations will truncate their operands.
13239 ConstantSDNode *Const =
13240 isConstOrConstSplat(XOR->getOperand(1), /*AllowUndefs*/ false,
13241 /*AllowTruncation*/ true);
13242 if (!Const || !Const->isOne())
13243 return SDValue();
13244
13245 // Rewrite into vselect(cond, rhs, lhs).
13246 SDValue Cond = XOR->getOperand(0);
13247 SDValue LHS = N->getOperand(1);
13248 SDValue RHS = N->getOperand(2);
13249 EVT Type = N->getValueType(0);
13250 return DCI.DAG.getNode(ISD::VSELECT, SDLoc(N), Type, Cond, RHS, LHS);
13251}
13252
13253// Convert vsetcc([0,1,2,..], splat(n), ult) -> vctp n
13256 const ARMSubtarget *Subtarget) {
13257 SDValue Op0 = N->getOperand(0);
13258 SDValue Op1 = N->getOperand(1);
13259 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
13260 EVT VT = N->getValueType(0);
13261
13262 if (!Subtarget->hasMVEIntegerOps() ||
13264 return SDValue();
13265
13266 if (CC == ISD::SETUGT) {
13267 std::swap(Op0, Op1);
13268 CC = ISD::SETULT;
13269 }
13270
13271 if (CC != ISD::SETULT || VT.getScalarSizeInBits() != 1 ||
13273 return SDValue();
13274
13275 // Check first operand is BuildVector of 0,1,2,...
13276 for (unsigned I = 0; I < VT.getVectorNumElements(); I++) {
13277 if (!Op0.getOperand(I).isUndef() &&
13279 Op0.getConstantOperandVal(I) == I))
13280 return SDValue();
13281 }
13282
13283 // The second is a Splat of Op1S
13284 SDValue Op1S = DCI.DAG.getSplatValue(Op1);
13285 if (!Op1S)
13286 return SDValue();
13287
13288 unsigned Opc;
13289 switch (VT.getVectorNumElements()) {
13290 case 4:
13291 Opc = Intrinsic::arm_mve_vctp32;
13292 break;
13293 case 8:
13294 Opc = Intrinsic::arm_mve_vctp16;
13295 break;
13296 case 16:
13297 Opc = Intrinsic::arm_mve_vctp8;
13298 break;
13299 default:
13300 return SDValue();
13301 }
13302
13303 SDLoc DL(N);
13304 return DCI.DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, VT,
13305 DCI.DAG.getConstant(Opc, DL, MVT::i32),
13306 DCI.DAG.getZExtOrTrunc(Op1S, DL, MVT::i32));
13307}
13308
13309/// PerformADDECombine - Target-specific dag combine transform from
13310/// ARMISD::ADDC, ARMISD::ADDE, and ISD::MUL_LOHI to MLAL or
13311/// ARMISD::ADDC, ARMISD::ADDE and ARMISD::UMLAL to ARMISD::UMAAL
13314 const ARMSubtarget *Subtarget) {
13315 // Only ARM and Thumb2 support UMLAL/SMLAL.
13316 if (Subtarget->isThumb1Only())
13317 return PerformAddeSubeCombine(N, DCI, Subtarget);
13318
13319 // Only perform the checks after legalize when the pattern is available.
13320 if (DCI.isBeforeLegalize()) return SDValue();
13321
13322 return AddCombineTo64bitUMAAL(N, DCI, Subtarget);
13323}
13324
13325/// PerformADDCombineWithOperands - Try DAG combinations for an ADD with
13326/// operands N0 and N1. This is a helper for PerformADDCombine that is
13327/// called with the default operands, and if that fails, with commuted
13328/// operands.
13331 const ARMSubtarget *Subtarget){
13332 // Attempt to create vpadd for this add.
13333 if (SDValue Result = AddCombineToVPADD(N, N0, N1, DCI, Subtarget))
13334 return Result;
13335
13336 // Attempt to create vpaddl for this add.
13337 if (SDValue Result = AddCombineVUZPToVPADDL(N, N0, N1, DCI, Subtarget))
13338 return Result;
13339 if (SDValue Result = AddCombineBUILD_VECTORToVPADDL(N, N0, N1, DCI,
13340 Subtarget))
13341 return Result;
13342
13343 // fold (add (select cc, 0, c), x) -> (select cc, x, (add, x, c))
13344 if (N0.getNode()->hasOneUse())
13345 if (SDValue Result = combineSelectAndUse(N, N0, N1, DCI))
13346 return Result;
13347 return SDValue();
13348}
13349
13351 EVT VT = N->getValueType(0);
13352 SDValue N0 = N->getOperand(0);
13353 SDValue N1 = N->getOperand(1);
13354 SDLoc dl(N);
13355
13356 auto IsVecReduce = [](SDValue Op) {
13357 switch (Op.getOpcode()) {
13358 case ISD::VECREDUCE_ADD:
13359 case ARMISD::VADDVs:
13360 case ARMISD::VADDVu:
13361 case ARMISD::VMLAVs:
13362 case ARMISD::VMLAVu:
13363 return true;
13364 }
13365 return false;
13366 };
13367
13368 auto DistrubuteAddAddVecReduce = [&](SDValue N0, SDValue N1) {
13369 // Distribute add(X, add(vecreduce(Y), vecreduce(Z))) ->
13370 // add(add(X, vecreduce(Y)), vecreduce(Z))
13371 // to make better use of vaddva style instructions.
13372 if (VT == MVT::i32 && N1.getOpcode() == ISD::ADD && !IsVecReduce(N0) &&
13373 IsVecReduce(N1.getOperand(0)) && IsVecReduce(N1.getOperand(1)) &&
13374 !isa<ConstantSDNode>(N0) && N1->hasOneUse()) {
13375 SDValue Add0 = DAG.getNode(ISD::ADD, dl, VT, N0, N1.getOperand(0));
13376 return DAG.getNode(ISD::ADD, dl, VT, Add0, N1.getOperand(1));
13377 }
13378 // And turn add(add(A, reduce(B)), add(C, reduce(D))) ->
13379 // add(add(add(A, C), reduce(B)), reduce(D))
13380 if (VT == MVT::i32 && N0.getOpcode() == ISD::ADD &&
13381 N1.getOpcode() == ISD::ADD && N0->hasOneUse() && N1->hasOneUse()) {
13382 unsigned N0RedOp = 0;
13383 if (!IsVecReduce(N0.getOperand(N0RedOp))) {
13384 N0RedOp = 1;
13385 if (!IsVecReduce(N0.getOperand(N0RedOp)))
13386 return SDValue();
13387 }
13388
13389 unsigned N1RedOp = 0;
13390 if (!IsVecReduce(N1.getOperand(N1RedOp)))
13391 N1RedOp = 1;
13392 if (!IsVecReduce(N1.getOperand(N1RedOp)))
13393 return SDValue();
13394
13395 SDValue Add0 = DAG.getNode(ISD::ADD, dl, VT, N0.getOperand(1 - N0RedOp),
13396 N1.getOperand(1 - N1RedOp));
13397 SDValue Add1 =
13398 DAG.getNode(ISD::ADD, dl, VT, Add0, N0.getOperand(N0RedOp));
13399 return DAG.getNode(ISD::ADD, dl, VT, Add1, N1.getOperand(N1RedOp));
13400 }
13401 return SDValue();
13402 };
13403 if (SDValue R = DistrubuteAddAddVecReduce(N0, N1))
13404 return R;
13405 if (SDValue R = DistrubuteAddAddVecReduce(N1, N0))
13406 return R;
13407
13408 // Distribute add(vecreduce(load(Y)), vecreduce(load(Z)))
13409 // Or add(add(X, vecreduce(load(Y))), vecreduce(load(Z)))
13410 // by ascending load offsets. This can help cores prefetch if the order of
13411 // loads is more predictable.
13412 auto DistrubuteVecReduceLoad = [&](SDValue N0, SDValue N1, bool IsForward) {
13413 // Check if two reductions are known to load data where one is before/after
13414 // another. Return negative if N0 loads data before N1, positive if N1 is
13415 // before N0 and 0 otherwise if nothing is known.
13416 auto IsKnownOrderedLoad = [&](SDValue N0, SDValue N1) {
13417 // Look through to the first operand of a MUL, for the VMLA case.
13418 // Currently only looks at the first operand, in the hope they are equal.
13419 if (N0.getOpcode() == ISD::MUL)
13420 N0 = N0.getOperand(0);
13421 if (N1.getOpcode() == ISD::MUL)
13422 N1 = N1.getOperand(0);
13423
13424 // Return true if the two operands are loads to the same object and the
13425 // offset of the first is known to be less than the offset of the second.
13426 LoadSDNode *Load0 = dyn_cast<LoadSDNode>(N0);
13427 LoadSDNode *Load1 = dyn_cast<LoadSDNode>(N1);
13428 if (!Load0 || !Load1 || Load0->getChain() != Load1->getChain() ||
13429 !Load0->isSimple() || !Load1->isSimple() || Load0->isIndexed() ||
13430 Load1->isIndexed())
13431 return 0;
13432
13433 auto BaseLocDecomp0 = BaseIndexOffset::match(Load0, DAG);
13434 auto BaseLocDecomp1 = BaseIndexOffset::match(Load1, DAG);
13435
13436 if (!BaseLocDecomp0.getBase() ||
13437 BaseLocDecomp0.getBase() != BaseLocDecomp1.getBase() ||
13438 !BaseLocDecomp0.hasValidOffset() || !BaseLocDecomp1.hasValidOffset())
13439 return 0;
13440 if (BaseLocDecomp0.getOffset() < BaseLocDecomp1.getOffset())
13441 return -1;
13442 if (BaseLocDecomp0.getOffset() > BaseLocDecomp1.getOffset())
13443 return 1;
13444 return 0;
13445 };
13446
13447 SDValue X;
13448 if (N0.getOpcode() == ISD::ADD && N0->hasOneUse()) {
13449 if (IsVecReduce(N0.getOperand(0)) && IsVecReduce(N0.getOperand(1))) {
13450 int IsBefore = IsKnownOrderedLoad(N0.getOperand(0).getOperand(0),
13451 N0.getOperand(1).getOperand(0));
13452 if (IsBefore < 0) {
13453 X = N0.getOperand(0);
13454 N0 = N0.getOperand(1);
13455 } else if (IsBefore > 0) {
13456 X = N0.getOperand(1);
13457 N0 = N0.getOperand(0);
13458 } else
13459 return SDValue();
13460 } else if (IsVecReduce(N0.getOperand(0))) {
13461 X = N0.getOperand(1);
13462 N0 = N0.getOperand(0);
13463 } else if (IsVecReduce(N0.getOperand(1))) {
13464 X = N0.getOperand(0);
13465 N0 = N0.getOperand(1);
13466 } else
13467 return SDValue();
13468 } else if (IsForward && IsVecReduce(N0) && IsVecReduce(N1) &&
13469 IsKnownOrderedLoad(N0.getOperand(0), N1.getOperand(0)) < 0) {
13470 // Note this is backward to how you would expect. We create
13471 // add(reduce(load + 16), reduce(load + 0)) so that the
13472 // add(reduce(load+16), X) is combined into VADDVA(X, load+16)), leaving
13473 // the X as VADDV(load + 0)
13474 return DAG.getNode(ISD::ADD, dl, VT, N1, N0);
13475 } else
13476 return SDValue();
13477
13478 if (!IsVecReduce(N0) || !IsVecReduce(N1))
13479 return SDValue();
13480
13481 if (IsKnownOrderedLoad(N1.getOperand(0), N0.getOperand(0)) >= 0)
13482 return SDValue();
13483
13484 // Switch from add(add(X, N0), N1) to add(add(X, N1), N0)
13485 SDValue Add0 = DAG.getNode(ISD::ADD, dl, VT, X, N1);
13486 return DAG.getNode(ISD::ADD, dl, VT, Add0, N0);
13487 };
13488 if (SDValue R = DistrubuteVecReduceLoad(N0, N1, true))
13489 return R;
13490 if (SDValue R = DistrubuteVecReduceLoad(N1, N0, false))
13491 return R;
13492 return SDValue();
13493}
13494
13496 const ARMSubtarget *Subtarget) {
13497 if (!Subtarget->hasMVEIntegerOps())
13498 return SDValue();
13499
13501 return R;
13502
13503 EVT VT = N->getValueType(0);
13504 SDValue N0 = N->getOperand(0);
13505 SDValue N1 = N->getOperand(1);
13506 SDLoc dl(N);
13507
13508 if (VT != MVT::i64)
13509 return SDValue();
13510
13511 // We are looking for a i64 add of a VADDLVx. Due to these being i64's, this
13512 // will look like:
13513 // t1: i32,i32 = ARMISD::VADDLVs x
13514 // t2: i64 = build_pair t1, t1:1
13515 // t3: i64 = add t2, y
13516 // Otherwise we try to push the add up above VADDLVAx, to potentially allow
13517 // the add to be simplified separately.
13518 // We also need to check for sext / zext and commutitive adds.
13519 auto MakeVecReduce = [&](unsigned Opcode, unsigned OpcodeA, SDValue NA,
13520 SDValue NB) {
13521 if (NB->getOpcode() != ISD::BUILD_PAIR)
13522 return SDValue();
13523 SDValue VecRed = NB->getOperand(0);
13524 if ((VecRed->getOpcode() != Opcode && VecRed->getOpcode() != OpcodeA) ||
13525 VecRed.getResNo() != 0 ||
13526 NB->getOperand(1) != SDValue(VecRed.getNode(), 1))
13527 return SDValue();
13528
13529 if (VecRed->getOpcode() == OpcodeA) {
13530 // add(NA, VADDLVA(Inp), Y) -> VADDLVA(add(NA, Inp), Y)
13531 SDValue Inp = DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64,
13532 VecRed.getOperand(0), VecRed.getOperand(1));
13533 NA = DAG.getNode(ISD::ADD, dl, MVT::i64, Inp, NA);
13534 }
13535
13537 std::tie(Ops[0], Ops[1]) = DAG.SplitScalar(NA, dl, MVT::i32, MVT::i32);
13538
13539 unsigned S = VecRed->getOpcode() == OpcodeA ? 2 : 0;
13540 for (unsigned I = S, E = VecRed.getNumOperands(); I < E; I++)
13541 Ops.push_back(VecRed->getOperand(I));
13542 SDValue Red =
13543 DAG.getNode(OpcodeA, dl, DAG.getVTList({MVT::i32, MVT::i32}), Ops);
13544 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Red,
13545 SDValue(Red.getNode(), 1));
13546 };
13547
13548 if (SDValue M = MakeVecReduce(ARMISD::VADDLVs, ARMISD::VADDLVAs, N0, N1))
13549 return M;
13550 if (SDValue M = MakeVecReduce(ARMISD::VADDLVu, ARMISD::VADDLVAu, N0, N1))
13551 return M;
13552 if (SDValue M = MakeVecReduce(ARMISD::VADDLVs, ARMISD::VADDLVAs, N1, N0))
13553 return M;
13554 if (SDValue M = MakeVecReduce(ARMISD::VADDLVu, ARMISD::VADDLVAu, N1, N0))
13555 return M;
13556 if (SDValue M = MakeVecReduce(ARMISD::VADDLVps, ARMISD::VADDLVAps, N0, N1))
13557 return M;
13558 if (SDValue M = MakeVecReduce(ARMISD::VADDLVpu, ARMISD::VADDLVApu, N0, N1))
13559 return M;
13560 if (SDValue M = MakeVecReduce(ARMISD::VADDLVps, ARMISD::VADDLVAps, N1, N0))
13561 return M;
13562 if (SDValue M = MakeVecReduce(ARMISD::VADDLVpu, ARMISD::VADDLVApu, N1, N0))
13563 return M;
13564 if (SDValue M = MakeVecReduce(ARMISD::VMLALVs, ARMISD::VMLALVAs, N0, N1))
13565 return M;
13566 if (SDValue M = MakeVecReduce(ARMISD::VMLALVu, ARMISD::VMLALVAu, N0, N1))
13567 return M;
13568 if (SDValue M = MakeVecReduce(ARMISD::VMLALVs, ARMISD::VMLALVAs, N1, N0))
13569 return M;
13570 if (SDValue M = MakeVecReduce(ARMISD::VMLALVu, ARMISD::VMLALVAu, N1, N0))
13571 return M;
13572 if (SDValue M = MakeVecReduce(ARMISD::VMLALVps, ARMISD::VMLALVAps, N0, N1))
13573 return M;
13574 if (SDValue M = MakeVecReduce(ARMISD::VMLALVpu, ARMISD::VMLALVApu, N0, N1))
13575 return M;
13576 if (SDValue M = MakeVecReduce(ARMISD::VMLALVps, ARMISD::VMLALVAps, N1, N0))
13577 return M;
13578 if (SDValue M = MakeVecReduce(ARMISD::VMLALVpu, ARMISD::VMLALVApu, N1, N0))
13579 return M;
13580 return SDValue();
13581}
13582
13583bool
13585 CombineLevel Level) const {
13586 assert((N->getOpcode() == ISD::SHL || N->getOpcode() == ISD::SRA ||
13587 N->getOpcode() == ISD::SRL) &&
13588 "Expected shift op");
13589
13590 SDValue ShiftLHS = N->getOperand(0);
13591 if (!ShiftLHS->hasOneUse())
13592 return false;
13593
13594 if (ShiftLHS.getOpcode() == ISD::SIGN_EXTEND &&
13595 !ShiftLHS.getOperand(0)->hasOneUse())
13596 return false;
13597
13598 if (Level == BeforeLegalizeTypes)
13599 return true;
13600
13601 if (N->getOpcode() != ISD::SHL)
13602 return true;
13603
13604 if (Subtarget->isThumb1Only()) {
13605 // Avoid making expensive immediates by commuting shifts. (This logic
13606 // only applies to Thumb1 because ARM and Thumb2 immediates can be shifted
13607 // for free.)
13608 if (N->getOpcode() != ISD::SHL)
13609 return true;
13610 SDValue N1 = N->getOperand(0);
13611 if (N1->getOpcode() != ISD::ADD && N1->getOpcode() != ISD::AND &&
13612 N1->getOpcode() != ISD::OR && N1->getOpcode() != ISD::XOR)
13613 return true;
13614 if (auto *Const = dyn_cast<ConstantSDNode>(N1->getOperand(1))) {
13615 if (Const->getAPIntValue().ult(256))
13616 return false;
13617 if (N1->getOpcode() == ISD::ADD && Const->getAPIntValue().slt(0) &&
13618 Const->getAPIntValue().sgt(-256))
13619 return false;
13620 }
13621 return true;
13622 }
13623
13624 // Turn off commute-with-shift transform after legalization, so it doesn't
13625 // conflict with PerformSHLSimplify. (We could try to detect when
13626 // PerformSHLSimplify would trigger more precisely, but it isn't
13627 // really necessary.)
13628 return false;
13629}
13630
13632 const SDNode *N) const {
13633 assert(N->getOpcode() == ISD::XOR &&
13634 (N->getOperand(0).getOpcode() == ISD::SHL ||
13635 N->getOperand(0).getOpcode() == ISD::SRL) &&
13636 "Expected XOR(SHIFT) pattern");
13637
13638 // Only commute if the entire NOT mask is a hidden shifted mask.
13639 auto *XorC = dyn_cast<ConstantSDNode>(N->getOperand(1));
13640 auto *ShiftC = dyn_cast<ConstantSDNode>(N->getOperand(0).getOperand(1));
13641 if (XorC && ShiftC) {
13642 unsigned MaskIdx, MaskLen;
13643 if (XorC->getAPIntValue().isShiftedMask(MaskIdx, MaskLen)) {
13644 unsigned ShiftAmt = ShiftC->getZExtValue();
13645 unsigned BitWidth = N->getValueType(0).getScalarSizeInBits();
13646 if (N->getOperand(0).getOpcode() == ISD::SHL)
13647 return MaskIdx == ShiftAmt && MaskLen == (BitWidth - ShiftAmt);
13648 return MaskIdx == 0 && MaskLen == (BitWidth - ShiftAmt);
13649 }
13650 }
13651
13652 return false;
13653}
13654
13656 const SDNode *N) const {
13657 assert(((N->getOpcode() == ISD::SHL &&
13658 N->getOperand(0).getOpcode() == ISD::SRL) ||
13659 (N->getOpcode() == ISD::SRL &&
13660 N->getOperand(0).getOpcode() == ISD::SHL)) &&
13661 "Expected shift-shift mask");
13662
13663 if (!Subtarget->isThumb1Only())
13664 return true;
13665
13666 EVT VT = N->getValueType(0);
13667 if (VT.getScalarSizeInBits() > 32)
13668 return true;
13669
13670 return false;
13671}
13672
13674 unsigned BinOpcode, EVT VT, unsigned SelectOpcode, SDValue X,
13675 SDValue Y) const {
13676 return Subtarget->hasMVEIntegerOps() && isTypeLegal(VT) &&
13677 SelectOpcode == ISD::VSELECT;
13678}
13679
13681 if (!Subtarget->hasNEON() && !Subtarget->hasMVEIntegerOps()) {
13682 if (Subtarget->isThumb1Only())
13683 return VT.getScalarSizeInBits() <= 32;
13684 return true;
13685 }
13686 return VT.isScalarInteger();
13687}
13688
13690 EVT VT) const {
13691 if (!isOperationLegalOrCustom(Op, VT) || !FPVT.isSimple())
13692 return false;
13693
13694 switch (FPVT.getSimpleVT().SimpleTy) {
13695 case MVT::f16:
13696 return Subtarget->hasVFP2Base();
13697 case MVT::f32:
13698 return Subtarget->hasVFP2Base();
13699 case MVT::f64:
13700 return Subtarget->hasFP64();
13701 case MVT::v4f32:
13702 case MVT::v8f16:
13703 return Subtarget->hasMVEFloatOps();
13704 default:
13705 return false;
13706 }
13707}
13708
13711 const ARMSubtarget *ST) {
13712 // Allow the generic combiner to identify potential bswaps.
13713 if (DCI.isBeforeLegalize())
13714 return SDValue();
13715
13716 // DAG combiner will fold:
13717 // (shl (add x, c1), c2) -> (add (shl x, c2), c1 << c2)
13718 // (shl (or x, c1), c2) -> (or (shl x, c2), c1 << c2
13719 // Other code patterns that can be also be modified have the following form:
13720 // b + ((a << 1) | 510)
13721 // b + ((a << 1) & 510)
13722 // b + ((a << 1) ^ 510)
13723 // b + ((a << 1) + 510)
13724
13725 // Many instructions can perform the shift for free, but it requires both
13726 // the operands to be registers. If c1 << c2 is too large, a mov immediate
13727 // instruction will needed. So, unfold back to the original pattern if:
13728 // - if c1 and c2 are small enough that they don't require mov imms.
13729 // - the user(s) of the node can perform an shl
13730
13731 // No shifted operands for 16-bit instructions.
13732 if (ST->isThumb1Only())
13733 return SDValue();
13734
13735 // Check that all the users could perform the shl themselves.
13736 for (auto *U : N->users()) {
13737 switch(U->getOpcode()) {
13738 default:
13739 return SDValue();
13740 case ISD::SUB:
13741 case ISD::ADD:
13742 case ISD::AND:
13743 case ISD::OR:
13744 case ISD::XOR:
13745 case ISD::SETCC:
13746 case ARMISD::CMP:
13747 // Check that the user isn't already using a constant because there
13748 // aren't any instructions that support an immediate operand and a
13749 // shifted operand.
13750 if (isa<ConstantSDNode>(U->getOperand(0)) ||
13751 isa<ConstantSDNode>(U->getOperand(1)))
13752 return SDValue();
13753
13754 // Check that it's not already using a shift.
13755 if (U->getOperand(0).getOpcode() == ISD::SHL ||
13756 U->getOperand(1).getOpcode() == ISD::SHL)
13757 return SDValue();
13758 break;
13759 }
13760 }
13761
13762 if (N->getOpcode() != ISD::ADD && N->getOpcode() != ISD::OR &&
13763 N->getOpcode() != ISD::XOR && N->getOpcode() != ISD::AND)
13764 return SDValue();
13765
13766 if (N->getOperand(0).getOpcode() != ISD::SHL)
13767 return SDValue();
13768
13769 SDValue SHL = N->getOperand(0);
13770
13771 auto *C1ShlC2 = dyn_cast<ConstantSDNode>(N->getOperand(1));
13772 auto *C2 = dyn_cast<ConstantSDNode>(SHL.getOperand(1));
13773 if (!C1ShlC2 || !C2)
13774 return SDValue();
13775
13776 APInt C2Int = C2->getAPIntValue();
13777 APInt C1Int = C1ShlC2->getAPIntValue();
13778 unsigned C2Width = C2Int.getBitWidth();
13779 if (C2Int.uge(C2Width))
13780 return SDValue();
13781 uint64_t C2Value = C2Int.getZExtValue();
13782
13783 // Check that performing a lshr will not lose any information.
13784 APInt Mask = APInt::getHighBitsSet(C2Width, C2Width - C2Value);
13785 if ((C1Int & Mask) != C1Int)
13786 return SDValue();
13787
13788 // Shift the first constant.
13789 C1Int.lshrInPlace(C2Int);
13790
13791 // The immediates are encoded as an 8-bit value that can be rotated.
13792 auto LargeImm = [](const APInt &Imm) {
13793 unsigned Zeros = Imm.countl_zero() + Imm.countr_zero();
13794 return Imm.getBitWidth() - Zeros > 8;
13795 };
13796
13797 if (LargeImm(C1Int) || LargeImm(C2Int))
13798 return SDValue();
13799
13800 SelectionDAG &DAG = DCI.DAG;
13801 SDLoc dl(N);
13802 SDValue X = SHL.getOperand(0);
13803 SDValue BinOp = DAG.getNode(N->getOpcode(), dl, MVT::i32, X,
13804 DAG.getConstant(C1Int, dl, MVT::i32));
13805 // Shift left to compensate for the lshr of C1Int.
13806 SDValue Res = DAG.getNode(ISD::SHL, dl, MVT::i32, BinOp, SHL.getOperand(1));
13807
13808 LLVM_DEBUG(dbgs() << "Simplify shl use:\n"; SHL.getOperand(0).dump();
13809 SHL.dump(); N->dump());
13810 LLVM_DEBUG(dbgs() << "Into:\n"; X.dump(); BinOp.dump(); Res.dump());
13811 return Res;
13812}
13813
13814
13815/// PerformADDCombine - Target-specific dag combine xforms for ISD::ADD.
13816///
13819 const ARMSubtarget *Subtarget) {
13820 SDValue N0 = N->getOperand(0);
13821 SDValue N1 = N->getOperand(1);
13822
13823 // Only works one way, because it needs an immediate operand.
13824 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget))
13825 return Result;
13826
13827 if (SDValue Result = PerformADDVecReduce(N, DCI.DAG, Subtarget))
13828 return Result;
13829
13830 // First try with the default operand order.
13831 if (SDValue Result = PerformADDCombineWithOperands(N, N0, N1, DCI, Subtarget))
13832 return Result;
13833
13834 // If that didn't work, try again with the operands commuted.
13835 return PerformADDCombineWithOperands(N, N1, N0, DCI, Subtarget);
13836}
13837
13838// Combine (sub 0, (csinc X, Y, CC)) -> (csinv -X, Y, CC)
13839// providing -X is as cheap as X (currently, just a constant).
13841 if (N->getValueType(0) != MVT::i32 || !isNullConstant(N->getOperand(0)))
13842 return SDValue();
13843 SDValue CSINC = N->getOperand(1);
13844 if (CSINC.getOpcode() != ARMISD::CSINC || !CSINC.hasOneUse())
13845 return SDValue();
13846
13848 if (!X)
13849 return SDValue();
13850
13851 return DAG.getNode(ARMISD::CSINV, SDLoc(N), MVT::i32,
13852 DAG.getNode(ISD::SUB, SDLoc(N), MVT::i32, N->getOperand(0),
13853 CSINC.getOperand(0)),
13854 CSINC.getOperand(1), CSINC.getOperand(2),
13855 CSINC.getOperand(3));
13856}
13857
13859 // Free to negate.
13861 return 0;
13862
13863 // Will save one instruction.
13864 if (Op.getOpcode() == ISD::SUB && isNullConstant(Op.getOperand(0)))
13865 return -1;
13866
13867 // Can freely negate by converting sra <-> srl.
13868 if (Op.getOpcode() == ISD::SRA || Op.getOpcode() == ISD::SRL) {
13869 ConstantSDNode *ShiftAmt = dyn_cast<ConstantSDNode>(Op.getOperand(1));
13870 if (Op.hasOneUse() && ShiftAmt &&
13871 ShiftAmt->getZExtValue() == Op.getValueType().getScalarSizeInBits() - 1)
13872 return 0;
13873 }
13874
13875 // Will have to create sub.
13876 return 1;
13877}
13878
13879// Try to fold
13880//
13881// (neg (cmov X, Y)) -> (cmov (neg X), (neg Y))
13882//
13883// The folding helps cmov to be matched with csneg without generating
13884// redundant neg instruction.
13886 assert(N->getOpcode() == ISD::SUB);
13887 if (!isNullConstant(N->getOperand(0)))
13888 return SDValue();
13889
13890 SDValue CMov = N->getOperand(1);
13891 if (CMov.getOpcode() != ARMISD::CMOV || !CMov->hasOneUse())
13892 return SDValue();
13893
13894 SDValue N0 = CMov.getOperand(0);
13895 SDValue N1 = CMov.getOperand(1);
13896
13897 // Only perform the fold if we actually save something.
13898 if (getNegationCost(N0) + getNegationCost(N1) > 0)
13899 return SDValue();
13900
13901 SDLoc DL(N);
13902 EVT VT = CMov.getValueType();
13903
13904 SDValue N0N = DAG.getNegative(N0, DL, VT);
13905 SDValue N1N = DAG.getNegative(N1, DL, VT);
13906 return DAG.getNode(ARMISD::CMOV, DL, VT, N0N, N1N, CMov.getOperand(2),
13907 CMov.getOperand(3));
13908}
13909
13910/// PerformSUBCombine - Target-specific dag combine xforms for ISD::SUB.
13911///
13914 const ARMSubtarget *Subtarget) {
13915 SDValue N0 = N->getOperand(0);
13916 SDValue N1 = N->getOperand(1);
13917
13918 // fold (sub x, (select cc, 0, c)) -> (select cc, x, (sub, x, c))
13919 if (N1.getNode()->hasOneUse())
13920 if (SDValue Result = combineSelectAndUse(N, N1, N0, DCI))
13921 return Result;
13922
13923 if (SDValue R = PerformSubCSINCCombine(N, DCI.DAG))
13924 return R;
13925
13926 if (SDValue Val = performNegCMovCombine(N, DCI.DAG))
13927 return Val;
13928
13929 if (!Subtarget->hasMVEIntegerOps() || !N->getValueType(0).isVector())
13930 return SDValue();
13931
13932 // Fold (sub (ARMvmovImm 0), (ARMvdup x)) -> (ARMvdup (sub 0, x))
13933 // so that we can readily pattern match more mve instructions which can use
13934 // a scalar operand.
13935 SDValue VDup = N->getOperand(1);
13936 if (VDup->getOpcode() != ARMISD::VDUP)
13937 return SDValue();
13938
13939 SDValue VMov = N->getOperand(0);
13940 if (VMov->getOpcode() == ISD::BITCAST)
13941 VMov = VMov->getOperand(0);
13942
13943 if (VMov->getOpcode() != ARMISD::VMOVIMM || !isZeroVector(VMov))
13944 return SDValue();
13945
13946 SDLoc dl(N);
13947 SDValue Negate = DCI.DAG.getNode(ISD::SUB, dl, MVT::i32,
13948 DCI.DAG.getConstant(0, dl, MVT::i32),
13949 VDup->getOperand(0));
13950 return DCI.DAG.getNode(ARMISD::VDUP, dl, N->getValueType(0), Negate);
13951}
13952
13953/// PerformVMULCombine
13954/// Distribute (A + B) * C to (A * C) + (B * C) to take advantage of the
13955/// special multiplier accumulator forwarding.
13956/// vmul d3, d0, d2
13957/// vmla d3, d1, d2
13958/// is faster than
13959/// vadd d3, d0, d1
13960/// vmul d3, d3, d2
13961// However, for (A + B) * (A + B),
13962// vadd d2, d0, d1
13963// vmul d3, d0, d2
13964// vmla d3, d1, d2
13965// is slower than
13966// vadd d2, d0, d1
13967// vmul d3, d2, d2
13970 const ARMSubtarget *Subtarget) {
13971 if (!Subtarget->hasVMLxForwarding())
13972 return SDValue();
13973
13974 SelectionDAG &DAG = DCI.DAG;
13975 SDValue N0 = N->getOperand(0);
13976 SDValue N1 = N->getOperand(1);
13977 unsigned Opcode = N0.getOpcode();
13978 if (Opcode != ISD::ADD && Opcode != ISD::SUB &&
13979 Opcode != ISD::FADD && Opcode != ISD::FSUB) {
13980 Opcode = N1.getOpcode();
13981 if (Opcode != ISD::ADD && Opcode != ISD::SUB &&
13982 Opcode != ISD::FADD && Opcode != ISD::FSUB)
13983 return SDValue();
13984 std::swap(N0, N1);
13985 }
13986
13987 if (N0 == N1)
13988 return SDValue();
13989
13990 EVT VT = N->getValueType(0);
13991 SDLoc DL(N);
13992 SDValue N00 = N0->getOperand(0);
13993 SDValue N01 = N0->getOperand(1);
13994 return DAG.getNode(Opcode, DL, VT,
13995 DAG.getNode(ISD::MUL, DL, VT, N00, N1),
13996 DAG.getNode(ISD::MUL, DL, VT, N01, N1));
13997}
13998
14000 const ARMSubtarget *Subtarget) {
14001 EVT VT = N->getValueType(0);
14002 if (VT != MVT::v2i64)
14003 return SDValue();
14004
14005 SDValue N0 = N->getOperand(0);
14006 SDValue N1 = N->getOperand(1);
14007
14008 auto IsSignExt = [&](SDValue Op) {
14009 if (Op->getOpcode() != ISD::SIGN_EXTEND_INREG)
14010 return SDValue();
14011 EVT VT = cast<VTSDNode>(Op->getOperand(1))->getVT();
14012 if (VT.getScalarSizeInBits() == 32)
14013 return Op->getOperand(0);
14014 return SDValue();
14015 };
14016 auto IsZeroExt = [&](SDValue Op) {
14017 // Zero extends are a little more awkward. At the point we are matching
14018 // this, we are looking for an AND with a (-1, 0, -1, 0) buildvector mask.
14019 // That might be before of after a bitcast depending on how the and is
14020 // placed. Because this has to look through bitcasts, it is currently only
14021 // supported on LE.
14022 if (!Subtarget->isLittle())
14023 return SDValue();
14024
14025 SDValue And = Op;
14026 if (And->getOpcode() == ISD::BITCAST)
14027 And = And->getOperand(0);
14028 if (And->getOpcode() != ISD::AND)
14029 return SDValue();
14030 SDValue Mask = And->getOperand(1);
14031 if (Mask->getOpcode() == ISD::BITCAST)
14032 Mask = Mask->getOperand(0);
14033
14034 if (Mask->getOpcode() != ISD::BUILD_VECTOR ||
14035 Mask.getValueType() != MVT::v4i32)
14036 return SDValue();
14037 if (isAllOnesConstant(Mask->getOperand(0)) &&
14038 isNullConstant(Mask->getOperand(1)) &&
14039 isAllOnesConstant(Mask->getOperand(2)) &&
14040 isNullConstant(Mask->getOperand(3)))
14041 return And->getOperand(0);
14042 return SDValue();
14043 };
14044
14045 SDLoc dl(N);
14046 if (SDValue Op0 = IsSignExt(N0)) {
14047 if (SDValue Op1 = IsSignExt(N1)) {
14048 SDValue New0a = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, Op0);
14049 SDValue New1a = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, Op1);
14050 return DAG.getNode(ARMISD::VMULLs, dl, VT, New0a, New1a);
14051 }
14052 }
14053 if (SDValue Op0 = IsZeroExt(N0)) {
14054 if (SDValue Op1 = IsZeroExt(N1)) {
14055 SDValue New0a = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, Op0);
14056 SDValue New1a = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v4i32, Op1);
14057 return DAG.getNode(ARMISD::VMULLu, dl, VT, New0a, New1a);
14058 }
14059 }
14060
14061 return SDValue();
14062}
14063
14066 const ARMSubtarget *Subtarget) {
14067 SelectionDAG &DAG = DCI.DAG;
14068
14069 EVT VT = N->getValueType(0);
14070 if (Subtarget->hasMVEIntegerOps() && VT == MVT::v2i64)
14071 return PerformMVEVMULLCombine(N, DAG, Subtarget);
14072
14073 if (Subtarget->isThumb1Only())
14074 return SDValue();
14075
14076 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer())
14077 return SDValue();
14078
14079 if (VT.is64BitVector() || VT.is128BitVector())
14080 return PerformVMULCombine(N, DCI, Subtarget);
14081 if (VT != MVT::i32)
14082 return SDValue();
14083
14084 ConstantSDNode *C = dyn_cast<ConstantSDNode>(N->getOperand(1));
14085 if (!C)
14086 return SDValue();
14087
14088 int64_t MulAmt = C->getSExtValue();
14089 unsigned ShiftAmt = llvm::countr_zero<uint64_t>(MulAmt);
14090
14091 ShiftAmt = ShiftAmt & (32 - 1);
14092 SDValue V = N->getOperand(0);
14093 SDLoc DL(N);
14094
14095 SDValue Res;
14096 MulAmt >>= ShiftAmt;
14097
14098 if (MulAmt >= 0) {
14099 if (llvm::has_single_bit<uint32_t>(MulAmt - 1)) {
14100 // (mul x, 2^N + 1) => (add (shl x, N), x)
14101 Res = DAG.getNode(ISD::ADD, DL, VT,
14102 V,
14103 DAG.getNode(ISD::SHL, DL, VT,
14104 V,
14105 DAG.getConstant(Log2_32(MulAmt - 1), DL,
14106 MVT::i32)));
14107 } else if (llvm::has_single_bit<uint32_t>(MulAmt + 1)) {
14108 // (mul x, 2^N - 1) => (sub (shl x, N), x)
14109 Res = DAG.getNode(ISD::SUB, DL, VT,
14110 DAG.getNode(ISD::SHL, DL, VT,
14111 V,
14112 DAG.getConstant(Log2_32(MulAmt + 1), DL,
14113 MVT::i32)),
14114 V);
14115 } else
14116 return SDValue();
14117 } else {
14118 uint64_t MulAmtAbs = -MulAmt;
14119 if (llvm::has_single_bit<uint32_t>(MulAmtAbs + 1)) {
14120 // (mul x, -(2^N - 1)) => (sub x, (shl x, N))
14121 Res = DAG.getNode(ISD::SUB, DL, VT,
14122 V,
14123 DAG.getNode(ISD::SHL, DL, VT,
14124 V,
14125 DAG.getConstant(Log2_32(MulAmtAbs + 1), DL,
14126 MVT::i32)));
14127 } else if (llvm::has_single_bit<uint32_t>(MulAmtAbs - 1)) {
14128 // (mul x, -(2^N + 1)) => - (add (shl x, N), x)
14129 Res = DAG.getNode(ISD::ADD, DL, VT,
14130 V,
14131 DAG.getNode(ISD::SHL, DL, VT,
14132 V,
14133 DAG.getConstant(Log2_32(MulAmtAbs - 1), DL,
14134 MVT::i32)));
14135 Res = DAG.getNode(ISD::SUB, DL, VT,
14136 DAG.getConstant(0, DL, MVT::i32), Res);
14137 } else
14138 return SDValue();
14139 }
14140
14141 if (ShiftAmt != 0)
14142 Res = DAG.getNode(ISD::SHL, DL, VT,
14143 Res, DAG.getConstant(ShiftAmt, DL, MVT::i32));
14144
14145 // Do not add new nodes to DAG combiner worklist.
14146 DCI.CombineTo(N, Res, false);
14147 return SDValue();
14148}
14149
14152 const ARMSubtarget *Subtarget) {
14153 // Allow DAGCombine to pattern-match before we touch the canonical form.
14154 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer())
14155 return SDValue();
14156
14157 if (N->getValueType(0) != MVT::i32)
14158 return SDValue();
14159
14160 ConstantSDNode *N1C = dyn_cast<ConstantSDNode>(N->getOperand(1));
14161 if (!N1C)
14162 return SDValue();
14163
14164 uint32_t C1 = (uint32_t)N1C->getZExtValue();
14165 // Don't transform uxtb/uxth.
14166 if (C1 == 255 || C1 == 65535)
14167 return SDValue();
14168
14169 SDNode *N0 = N->getOperand(0).getNode();
14170 if (!N0->hasOneUse())
14171 return SDValue();
14172
14173 if (N0->getOpcode() != ISD::SHL && N0->getOpcode() != ISD::SRL)
14174 return SDValue();
14175
14176 bool LeftShift = N0->getOpcode() == ISD::SHL;
14177
14179 if (!N01C)
14180 return SDValue();
14181
14182 uint32_t C2 = (uint32_t)N01C->getZExtValue();
14183 if (!C2 || C2 >= 32)
14184 return SDValue();
14185
14186 // Clear irrelevant bits in the mask.
14187 if (LeftShift)
14188 C1 &= (-1U << C2);
14189 else
14190 C1 &= (-1U >> C2);
14191
14192 SelectionDAG &DAG = DCI.DAG;
14193 SDLoc DL(N);
14194
14195 // We have a pattern of the form "(and (shl x, c2) c1)" or
14196 // "(and (srl x, c2) c1)", where c1 is a shifted mask. Try to
14197 // transform to a pair of shifts, to save materializing c1.
14198
14199 // First pattern: right shift, then mask off leading bits.
14200 // FIXME: Use demanded bits?
14201 if (!LeftShift && isMask_32(C1)) {
14202 uint32_t C3 = llvm::countl_zero(C1);
14203 if (C2 < C3) {
14204 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0),
14205 DAG.getConstant(C3 - C2, DL, MVT::i32));
14206 return DAG.getNode(ISD::SRL, DL, MVT::i32, SHL,
14207 DAG.getConstant(C3, DL, MVT::i32));
14208 }
14209 }
14210
14211 // First pattern, reversed: left shift, then mask off trailing bits.
14212 if (LeftShift && isMask_32(~C1)) {
14213 uint32_t C3 = llvm::countr_zero(C1);
14214 if (C2 < C3) {
14215 SDValue SHL = DAG.getNode(ISD::SRL, DL, MVT::i32, N0->getOperand(0),
14216 DAG.getConstant(C3 - C2, DL, MVT::i32));
14217 return DAG.getNode(ISD::SHL, DL, MVT::i32, SHL,
14218 DAG.getConstant(C3, DL, MVT::i32));
14219 }
14220 }
14221
14222 // Second pattern: left shift, then mask off leading bits.
14223 // FIXME: Use demanded bits?
14224 if (LeftShift && isShiftedMask_32(C1)) {
14225 uint32_t Trailing = llvm::countr_zero(C1);
14226 uint32_t C3 = llvm::countl_zero(C1);
14227 if (Trailing == C2 && C2 + C3 < 32) {
14228 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0),
14229 DAG.getConstant(C2 + C3, DL, MVT::i32));
14230 return DAG.getNode(ISD::SRL, DL, MVT::i32, SHL,
14231 DAG.getConstant(C3, DL, MVT::i32));
14232 }
14233 }
14234
14235 // Second pattern, reversed: right shift, then mask off trailing bits.
14236 // FIXME: Handle other patterns of known/demanded bits.
14237 if (!LeftShift && isShiftedMask_32(C1)) {
14238 uint32_t Leading = llvm::countl_zero(C1);
14239 uint32_t C3 = llvm::countr_zero(C1);
14240 if (Leading == C2 && C2 + C3 < 32) {
14241 SDValue SHL = DAG.getNode(ISD::SRL, DL, MVT::i32, N0->getOperand(0),
14242 DAG.getConstant(C2 + C3, DL, MVT::i32));
14243 return DAG.getNode(ISD::SHL, DL, MVT::i32, SHL,
14244 DAG.getConstant(C3, DL, MVT::i32));
14245 }
14246 }
14247
14248 // Transform "(and (shl x, c2) c1)" into "(shl (and x, c1>>c2), c2)"
14249 // if "c1 >> c2" is a cheaper immediate than "c1"
14250 if (LeftShift &&
14251 HasLowerConstantMaterializationCost(C1 >> C2, C1, Subtarget)) {
14252
14253 SDValue And = DAG.getNode(ISD::AND, DL, MVT::i32, N0->getOperand(0),
14254 DAG.getConstant(C1 >> C2, DL, MVT::i32));
14255 return DAG.getNode(ISD::SHL, DL, MVT::i32, And,
14256 DAG.getConstant(C2, DL, MVT::i32));
14257 }
14258
14259 return SDValue();
14260}
14261
14264 const ARMSubtarget *Subtarget) {
14265 // Attempt to use immediate-form VBIC
14266 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(N->getOperand(1));
14267 SDLoc dl(N);
14268 EVT VT = N->getValueType(0);
14269 SelectionDAG &DAG = DCI.DAG;
14270
14271 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT) || VT == MVT::v2i1 ||
14272 VT == MVT::v4i1 || VT == MVT::v8i1 || VT == MVT::v16i1)
14273 return SDValue();
14274
14275 APInt SplatBits, SplatUndef;
14276 unsigned SplatBitSize;
14277 bool HasAnyUndefs;
14278 if (BVN && (Subtarget->hasNEON() || Subtarget->hasMVEIntegerOps()) &&
14279 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) {
14280 if (SplatBitSize == 8 || SplatBitSize == 16 || SplatBitSize == 32 ||
14281 SplatBitSize == 64) {
14282 EVT VbicVT;
14283 SDValue Val = isVMOVModifiedImm((~SplatBits).getZExtValue(),
14284 SplatUndef.getZExtValue(), SplatBitSize,
14285 DAG, dl, VbicVT, VT, OtherModImm);
14286 if (Val.getNode()) {
14287 SDValue Input =
14288 DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VbicVT, N->getOperand(0));
14289 SDValue Vbic = DAG.getNode(ARMISD::VBICIMM, dl, VbicVT, Input, Val);
14290 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Vbic);
14291 }
14292 }
14293 }
14294
14295 if (!Subtarget->isThumb1Only()) {
14296 // fold (and (select cc, -1, c), x) -> (select cc, x, (and, x, c))
14297 if (SDValue Result = combineSelectAndUseCommutative(N, true, DCI))
14298 return Result;
14299
14300 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget))
14301 return Result;
14302 }
14303
14304 if (Subtarget->isThumb1Only())
14305 if (SDValue Result = CombineANDShift(N, DCI, Subtarget))
14306 return Result;
14307
14308 return SDValue();
14309}
14310
14311// Try combining OR nodes to SMULWB, SMULWT.
14314 const ARMSubtarget *Subtarget) {
14315 if (!Subtarget->hasV6Ops() ||
14316 (Subtarget->isThumb() &&
14317 (!Subtarget->hasThumb2() || !Subtarget->hasDSP())))
14318 return SDValue();
14319
14320 SDValue SRL = OR->getOperand(0);
14321 SDValue SHL = OR->getOperand(1);
14322
14323 if (SRL.getOpcode() != ISD::SRL || SHL.getOpcode() != ISD::SHL) {
14324 SRL = OR->getOperand(1);
14325 SHL = OR->getOperand(0);
14326 }
14327 if (!isSRL16(SRL) || !isSHL16(SHL))
14328 return SDValue();
14329
14330 // The first operands to the shifts need to be the two results from the
14331 // same smul_lohi node.
14332 if ((SRL.getOperand(0).getNode() != SHL.getOperand(0).getNode()) ||
14333 SRL.getOperand(0).getOpcode() != ISD::SMUL_LOHI)
14334 return SDValue();
14335
14336 SDNode *SMULLOHI = SRL.getOperand(0).getNode();
14337 if (SRL.getOperand(0) != SDValue(SMULLOHI, 0) ||
14338 SHL.getOperand(0) != SDValue(SMULLOHI, 1))
14339 return SDValue();
14340
14341 // Now we have:
14342 // (or (srl (smul_lohi ?, ?), 16), (shl (smul_lohi ?, ?), 16)))
14343 // For SMUL[B|T] smul_lohi will take a 32-bit and a 16-bit arguments.
14344 // For SMUWB the 16-bit value will signed extended somehow.
14345 // For SMULWT only the SRA is required.
14346 // Check both sides of SMUL_LOHI
14347 SDValue OpS16 = SMULLOHI->getOperand(0);
14348 SDValue OpS32 = SMULLOHI->getOperand(1);
14349
14350 SelectionDAG &DAG = DCI.DAG;
14351 if (!isS16(OpS16, DAG) && !isSRA16(OpS16)) {
14352 OpS16 = OpS32;
14353 OpS32 = SMULLOHI->getOperand(0);
14354 }
14355
14356 SDLoc dl(OR);
14357 unsigned Opcode = 0;
14358 if (isS16(OpS16, DAG))
14359 Opcode = ARMISD::SMULWB;
14360 else if (isSRA16(OpS16)) {
14361 Opcode = ARMISD::SMULWT;
14362 OpS16 = OpS16->getOperand(0);
14363 }
14364 else
14365 return SDValue();
14366
14367 SDValue Res = DAG.getNode(Opcode, dl, MVT::i32, OpS32, OpS16);
14368 DAG.ReplaceAllUsesOfValueWith(SDValue(OR, 0), Res);
14369 return SDValue(OR, 0);
14370}
14371
14374 const ARMSubtarget *Subtarget) {
14375 // BFI is only available on V6T2+
14376 if (Subtarget->isThumb1Only() || !Subtarget->hasV6T2Ops())
14377 return SDValue();
14378
14379 EVT VT = N->getValueType(0);
14380 SDValue N0 = N->getOperand(0);
14381 SDValue N1 = N->getOperand(1);
14382 SelectionDAG &DAG = DCI.DAG;
14383 SDLoc DL(N);
14384 // 1) or (and A, mask), val => ARMbfi A, val, mask
14385 // iff (val & mask) == val
14386 //
14387 // 2) or (and A, mask), (and B, mask2) => ARMbfi A, (lsr B, amt), mask
14388 // 2a) iff isBitFieldInvertedMask(mask) && isBitFieldInvertedMask(~mask2)
14389 // && mask == ~mask2
14390 // 2b) iff isBitFieldInvertedMask(~mask) && isBitFieldInvertedMask(mask2)
14391 // && ~mask == mask2
14392 // (i.e., copy a bitfield value into another bitfield of the same width)
14393
14394 if (VT != MVT::i32)
14395 return SDValue();
14396
14397 SDValue N00 = N0.getOperand(0);
14398
14399 // The value and the mask need to be constants so we can verify this is
14400 // actually a bitfield set. If the mask is 0xffff, we can do better
14401 // via a movt instruction, so don't use BFI in that case.
14402 SDValue MaskOp = N0.getOperand(1);
14404 if (!MaskC)
14405 return SDValue();
14406 unsigned Mask = MaskC->getZExtValue();
14407 if (Mask == 0xffff)
14408 return SDValue();
14409 SDValue Res;
14410 // Case (1): or (and A, mask), val => ARMbfi A, val, mask
14412 if (N1C) {
14413 unsigned Val = N1C->getZExtValue();
14414 if ((Val & ~Mask) != Val)
14415 return SDValue();
14416
14417 if (ARM::isBitFieldInvertedMask(Mask)) {
14418 Val >>= llvm::countr_zero(~Mask);
14419
14420 Res = DAG.getNode(ARMISD::BFI, DL, VT, N00,
14421 DAG.getConstant(Val, DL, MVT::i32),
14422 DAG.getConstant(Mask, DL, MVT::i32));
14423
14424 DCI.CombineTo(N, Res, false);
14425 // Return value from the original node to inform the combiner than N is
14426 // now dead.
14427 return SDValue(N, 0);
14428 }
14429 } else if (N1.getOpcode() == ISD::AND) {
14430 // case (2) or (and A, mask), (and B, mask2) => ARMbfi A, (lsr B, amt), mask
14432 if (!N11C)
14433 return SDValue();
14434 unsigned Mask2 = N11C->getZExtValue();
14435
14436 // Mask and ~Mask2 (or reverse) must be equivalent for the BFI pattern
14437 // as is to match.
14438 if (ARM::isBitFieldInvertedMask(Mask) &&
14439 (Mask == ~Mask2)) {
14440 // The pack halfword instruction works better for masks that fit it,
14441 // so use that when it's available.
14442 if (Subtarget->hasDSP() &&
14443 (Mask == 0xffff || Mask == 0xffff0000))
14444 return SDValue();
14445 // 2a
14446 unsigned amt = llvm::countr_zero(Mask2);
14447 Res = DAG.getNode(ISD::SRL, DL, VT, N1.getOperand(0),
14448 DAG.getConstant(amt, DL, MVT::i32));
14449 Res = DAG.getNode(ARMISD::BFI, DL, VT, N00, Res,
14450 DAG.getConstant(Mask, DL, MVT::i32));
14451 DCI.CombineTo(N, Res, false);
14452 // Return value from the original node to inform the combiner than N is
14453 // now dead.
14454 return SDValue(N, 0);
14455 } else if (ARM::isBitFieldInvertedMask(~Mask) &&
14456 (~Mask == Mask2)) {
14457 // The pack halfword instruction works better for masks that fit it,
14458 // so use that when it's available.
14459 if (Subtarget->hasDSP() &&
14460 (Mask2 == 0xffff || Mask2 == 0xffff0000))
14461 return SDValue();
14462 // 2b
14463 unsigned lsb = llvm::countr_zero(Mask);
14464 Res = DAG.getNode(ISD::SRL, DL, VT, N00,
14465 DAG.getConstant(lsb, DL, MVT::i32));
14466 Res = DAG.getNode(ARMISD::BFI, DL, VT, N1.getOperand(0), Res,
14467 DAG.getConstant(Mask2, DL, MVT::i32));
14468 DCI.CombineTo(N, Res, false);
14469 // Return value from the original node to inform the combiner than N is
14470 // now dead.
14471 return SDValue(N, 0);
14472 }
14473 }
14474
14475 if (DAG.MaskedValueIsZero(N1, MaskC->getAPIntValue()) &&
14476 N00.getOpcode() == ISD::SHL && isa<ConstantSDNode>(N00.getOperand(1)) &&
14478 // Case (3): or (and (shl A, #shamt), mask), B => ARMbfi B, A, ~mask
14479 // where lsb(mask) == #shamt and masked bits of B are known zero.
14480 SDValue ShAmt = N00.getOperand(1);
14481 unsigned ShAmtC = ShAmt->getAsZExtVal();
14482 unsigned LSB = llvm::countr_zero(Mask);
14483 if (ShAmtC != LSB)
14484 return SDValue();
14485
14486 Res = DAG.getNode(ARMISD::BFI, DL, VT, N1, N00.getOperand(0),
14487 DAG.getConstant(~Mask, DL, MVT::i32));
14488
14489 DCI.CombineTo(N, Res, false);
14490 // Return value from the original node to inform the combiner than N is
14491 // now dead.
14492 return SDValue(N, 0);
14493 }
14494
14495 return SDValue();
14496}
14497
14498static bool isValidMVECond(unsigned CC, bool IsFloat) {
14499 switch (CC) {
14500 case ARMCC::EQ:
14501 case ARMCC::NE:
14502 case ARMCC::LE:
14503 case ARMCC::GT:
14504 case ARMCC::GE:
14505 case ARMCC::LT:
14506 return true;
14507 case ARMCC::HS:
14508 case ARMCC::HI:
14509 return !IsFloat;
14510 default:
14511 return false;
14512 };
14513}
14514
14516 if (N->getOpcode() == ARMISD::VCMP)
14517 return (ARMCC::CondCodes)N->getConstantOperandVal(2);
14518 else if (N->getOpcode() == ARMISD::VCMPZ)
14519 return (ARMCC::CondCodes)N->getConstantOperandVal(1);
14520 else
14521 llvm_unreachable("Not a VCMP/VCMPZ!");
14522}
14523
14526 return isValidMVECond(CC, N->getOperand(0).getValueType().isFloatingPoint());
14527}
14528
14530 const ARMSubtarget *Subtarget) {
14531 // Try to invert "or A, B" -> "and ~A, ~B", as the "and" is easier to chain
14532 // together with predicates
14533 EVT VT = N->getValueType(0);
14534 SDLoc DL(N);
14535 SDValue N0 = N->getOperand(0);
14536 SDValue N1 = N->getOperand(1);
14537
14538 auto IsFreelyInvertable = [&](SDValue V) {
14539 if (V->getOpcode() == ARMISD::VCMP || V->getOpcode() == ARMISD::VCMPZ)
14540 return CanInvertMVEVCMP(V);
14541 return false;
14542 };
14543
14544 // At least one operand must be freely invertable.
14545 if (!(IsFreelyInvertable(N0) || IsFreelyInvertable(N1)))
14546 return SDValue();
14547
14548 SDValue NewN0 = DAG.getLogicalNOT(DL, N0, VT);
14549 SDValue NewN1 = DAG.getLogicalNOT(DL, N1, VT);
14550 SDValue And = DAG.getNode(ISD::AND, DL, VT, NewN0, NewN1);
14551 return DAG.getLogicalNOT(DL, And, VT);
14552}
14553
14554// Try to form a NEON shift-{right, left}-and-insert (VSRI/VSLI) from:
14555// (or (and X, splat (i32 C1)), (srl Y, splat (i32 C2))) -> VSRI X, Y, #C2
14556// (or (and X, splat (i32 C1)), (shl Y, splat (i32 C2))) -> VSLI X, Y, #C2
14557// where C1 is a mask that preserves the bits not written by the shift/insert,
14558// i.e. `C1 == (1 << C2) - 1`.
14560 SDValue ShiftOp, EVT VT,
14561 SDLoc dl) {
14562 // Match (and X, Mask)
14563 if (AndOp.getOpcode() != ISD::AND)
14564 return SDValue();
14565
14566 SDValue X = AndOp.getOperand(0);
14567 SDValue Mask = AndOp.getOperand(1);
14568
14569 ConstantSDNode *MaskC = isConstOrConstSplat(Mask, false, true);
14570 if (!MaskC)
14571 return SDValue();
14572 APInt MaskBits =
14573 MaskC->getAPIntValue().trunc(Mask.getScalarValueSizeInBits());
14574
14575 // Match shift (srl/shl Y, CntVec)
14576 int64_t Cnt = 0;
14577 bool IsShiftRight = false;
14578 SDValue Y;
14579
14580 if (ShiftOp.getOpcode() == ARMISD::VSHRuIMM) {
14581 IsShiftRight = true;
14582 Y = ShiftOp.getOperand(0);
14583 Cnt = ShiftOp.getConstantOperandVal(1);
14584 } else if (ShiftOp.getOpcode() == ARMISD::VSHLIMM) {
14585 Y = ShiftOp.getOperand(0);
14586 Cnt = ShiftOp.getConstantOperandVal(1);
14587 } else {
14588 return SDValue();
14589 }
14590
14591 unsigned ElemBits = VT.getScalarSizeInBits();
14592 APInt RequiredMask = IsShiftRight
14593 ? APInt::getHighBitsSet(ElemBits, (unsigned)Cnt)
14594 : APInt::getLowBitsSet(ElemBits, (unsigned)Cnt);
14595 if (MaskBits != RequiredMask)
14596 return SDValue();
14597
14598 unsigned Opc = IsShiftRight ? ARMISD::VSRIIMM : ARMISD::VSLIIMM;
14599 return DAG.getNode(Opc, dl, VT, X, Y, DAG.getConstant(Cnt, dl, MVT::i32));
14600}
14601
14602/// PerformORCombine - Target-specific dag combine xforms for ISD::OR
14604 const ARMSubtarget *Subtarget) {
14605 // Attempt to use immediate-form VORR
14606 BuildVectorSDNode *BVN = dyn_cast<BuildVectorSDNode>(N->getOperand(1));
14607 SDLoc dl(N);
14608 EVT VT = N->getValueType(0);
14609 SelectionDAG &DAG = DCI.DAG;
14610
14611 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
14612 return SDValue();
14613
14614 if (Subtarget->hasMVEIntegerOps() && (VT == MVT::v2i1 || VT == MVT::v4i1 ||
14615 VT == MVT::v8i1 || VT == MVT::v16i1))
14616 return PerformORCombine_i1(N, DAG, Subtarget);
14617
14618 APInt SplatBits, SplatUndef;
14619 unsigned SplatBitSize;
14620 bool HasAnyUndefs;
14621 if (BVN && (Subtarget->hasNEON() || Subtarget->hasMVEIntegerOps()) &&
14622 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs)) {
14623 if (SplatBitSize == 8 || SplatBitSize == 16 || SplatBitSize == 32 ||
14624 SplatBitSize == 64) {
14625 EVT VorrVT;
14626 SDValue Val =
14627 isVMOVModifiedImm(SplatBits.getZExtValue(), SplatUndef.getZExtValue(),
14628 SplatBitSize, DAG, dl, VorrVT, VT, OtherModImm);
14629 if (Val.getNode()) {
14630 SDValue Input =
14631 DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VorrVT, N->getOperand(0));
14632 SDValue Vorr = DAG.getNode(ARMISD::VORRIMM, dl, VorrVT, Input, Val);
14633 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Vorr);
14634 }
14635 }
14636 }
14637
14638 if (!Subtarget->isThumb1Only()) {
14639 // fold (or (select cc, 0, c), x) -> (select cc, x, (or, x, c))
14640 if (SDValue Result = combineSelectAndUseCommutative(N, false, DCI))
14641 return Result;
14642 if (SDValue Result = PerformORCombineToSMULWBT(N, DCI, Subtarget))
14643 return Result;
14644 }
14645
14646 SDValue N0 = N->getOperand(0);
14647 SDValue N1 = N->getOperand(1);
14648
14649 // (or (and X, C1), (srl Y, C2)) -> VSRI X, Y, #C2
14650 // (or (and X, C1), (shl Y, C2)) -> VSLI X, Y, #C2
14651 if (VT.isVector() &&
14652 ((Subtarget->hasNEON() && DAG.getTargetLoweringInfo().isTypeLegal(VT)) ||
14653 (Subtarget->hasMVEIntegerOps() &&
14654 (VT == MVT::v16i8 || VT == MVT::v8i16 || VT == MVT::v4i32)))) {
14655 if (SDValue ShiftInsert =
14656 PerformORCombineToShiftInsert(DAG, N0, N1, VT, dl))
14657 return ShiftInsert;
14658
14659 if (SDValue ShiftInsert =
14660 PerformORCombineToShiftInsert(DAG, N1, N0, VT, dl))
14661 return ShiftInsert;
14662 }
14663
14664 // (or (and B, A), (and C, ~A)) => (VBSL A, B, C) when A is a constant.
14665 if (Subtarget->hasNEON() && N1.getOpcode() == ISD::AND && VT.isVector() &&
14667
14668 // The code below optimizes (or (and X, Y), Z).
14669 // The AND operand needs to have a single user to make these optimizations
14670 // profitable.
14671 if (N0.getOpcode() != ISD::AND || !N0.hasOneUse())
14672 return SDValue();
14673
14674 APInt SplatUndef;
14675 unsigned SplatBitSize;
14676 bool HasAnyUndefs;
14677
14678 APInt SplatBits0, SplatBits1;
14681 // Ensure that the second operand of both ands are constants
14682 if (BVN0 && BVN0->isConstantSplat(SplatBits0, SplatUndef, SplatBitSize,
14683 HasAnyUndefs) && !HasAnyUndefs) {
14684 if (BVN1 && BVN1->isConstantSplat(SplatBits1, SplatUndef, SplatBitSize,
14685 HasAnyUndefs) && !HasAnyUndefs) {
14686 // Ensure that the bit width of the constants are the same and that
14687 // the splat arguments are logical inverses as per the pattern we
14688 // are trying to simplify.
14689 if (SplatBits0.getBitWidth() == SplatBits1.getBitWidth() &&
14690 SplatBits0 == ~SplatBits1) {
14691 // Canonicalize the vector type to make instruction selection
14692 // simpler.
14693 EVT CanonicalVT = VT.is128BitVector() ? MVT::v4i32 : MVT::v2i32;
14694 SDValue Mask = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl,
14695 CanonicalVT, N0->getOperand(1));
14696 SDValue LHS = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl,
14697 CanonicalVT, N0->getOperand(0));
14698 SDValue RHS = DAG.getNode(ARMISD::VECTOR_REG_CAST, dl,
14699 CanonicalVT, N1->getOperand(0));
14700 SDValue Result =
14701 DAG.getNode(ARMISD::VBSP, dl, CanonicalVT, Mask, LHS, RHS);
14702 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Result);
14703 }
14704 }
14705 }
14706 }
14707
14708 // Try to use the ARM/Thumb2 BFI (bitfield insert) instruction when
14709 // reasonable.
14710 if (N0.getOpcode() == ISD::AND && N0.hasOneUse()) {
14711 if (SDValue Res = PerformORCombineToBFI(N, DCI, Subtarget))
14712 return Res;
14713 }
14714
14715 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget))
14716 return Result;
14717
14718 // (or x, (csinc 0, 0, cc)) -> (csinc x, 0, cc)
14719 // providing that the x is 0 or 1.
14720 SDValue CSINC = N1;
14721 SDValue Other = N0;
14722 if (CSINC.getOpcode() != ARMISD::CSINC)
14723 std::swap(CSINC, Other);
14724 if (CSINC.getOpcode() == ARMISD::CSINC &&
14725 isNullConstant(CSINC.getOperand(0)) &&
14726 isNullConstant(CSINC.getOperand(1)) &&
14728 return DAG.getNode(ARMISD::CSINC, dl, VT, Other, CSINC.getOperand(1),
14729 CSINC.getOperand(2), CSINC.getOperand(3));
14730
14731 return SDValue();
14732}
14733
14736 const ARMSubtarget *Subtarget) {
14737 EVT VT = N->getValueType(0);
14738 SelectionDAG &DAG = DCI.DAG;
14739
14740 if(!DAG.getTargetLoweringInfo().isTypeLegal(VT))
14741 return SDValue();
14742
14743 if (!Subtarget->isThumb1Only()) {
14744 // fold (xor (select cc, 0, c), x) -> (select cc, x, (xor, x, c))
14745 if (SDValue Result = combineSelectAndUseCommutative(N, false, DCI))
14746 return Result;
14747
14748 if (SDValue Result = PerformSHLSimplify(N, DCI, Subtarget))
14749 return Result;
14750 }
14751
14752 if (Subtarget->hasMVEIntegerOps()) {
14753 // fold (xor(vcmp/z, 1)) into a vcmp with the opposite condition.
14754 SDValue N0 = N->getOperand(0);
14755 SDValue N1 = N->getOperand(1);
14756 const TargetLowering *TLI = Subtarget->getTargetLowering();
14757 if (TLI->isConstTrueVal(N1) &&
14758 (N0->getOpcode() == ARMISD::VCMP || N0->getOpcode() == ARMISD::VCMPZ)) {
14759 if (CanInvertMVEVCMP(N0)) {
14760 SDLoc DL(N0);
14762
14764 Ops.push_back(N0->getOperand(0));
14765 if (N0->getOpcode() == ARMISD::VCMP)
14766 Ops.push_back(N0->getOperand(1));
14767 Ops.push_back(DAG.getConstant(CC, DL, MVT::i32));
14768 return DAG.getNode(N0->getOpcode(), DL, N0->getValueType(0), Ops);
14769 }
14770 }
14771 }
14772
14773 return SDValue();
14774}
14775
14776// ParseBFI - given a BFI instruction in N, extract the "from" value (Rn) and return it,
14777// and fill in FromMask and ToMask with (consecutive) bits in "from" to be extracted and
14778// their position in "to" (Rd).
14779static SDValue ParseBFI(SDNode *N, APInt &ToMask, APInt &FromMask) {
14780 assert(N->getOpcode() == ARMISD::BFI);
14781
14782 SDValue From = N->getOperand(1);
14783 ToMask = ~N->getConstantOperandAPInt(2);
14784 FromMask = APInt::getLowBitsSet(ToMask.getBitWidth(), ToMask.popcount());
14785
14786 // If the Base came from a SHR #C, we can deduce that it is really testing bit
14787 // #C in the base of the SHR.
14788 if (From->getOpcode() == ISD::SRL &&
14789 isa<ConstantSDNode>(From->getOperand(1))) {
14790 APInt Shift = From->getConstantOperandAPInt(1);
14791 assert(Shift.getLimitedValue() < 32 && "Shift too large!");
14792 FromMask <<= Shift.getLimitedValue(31);
14793 From = From->getOperand(0);
14794 }
14795
14796 return From;
14797}
14798
14799// If A and B contain one contiguous set of bits, does A | B == A . B?
14800//
14801// Neither A nor B must be zero.
14802static bool BitsProperlyConcatenate(const APInt &A, const APInt &B) {
14803 unsigned LastActiveBitInA = A.countr_zero();
14804 unsigned FirstActiveBitInB = B.getBitWidth() - B.countl_zero() - 1;
14805 return LastActiveBitInA - 1 == FirstActiveBitInB;
14806}
14807
14809 // We have a BFI in N. Find a BFI it can combine with, if one exists.
14810 APInt ToMask, FromMask;
14811 SDValue From = ParseBFI(N, ToMask, FromMask);
14812 SDValue To = N->getOperand(0);
14813
14814 SDValue V = To;
14815 if (V.getOpcode() != ARMISD::BFI)
14816 return SDValue();
14817
14818 APInt NewToMask, NewFromMask;
14819 SDValue NewFrom = ParseBFI(V.getNode(), NewToMask, NewFromMask);
14820 if (NewFrom != From)
14821 return SDValue();
14822
14823 // Do the written bits conflict with any we've seen so far?
14824 if ((NewToMask & ToMask).getBoolValue())
14825 // Conflicting bits.
14826 return SDValue();
14827
14828 // Are the new bits contiguous when combined with the old bits?
14829 if (BitsProperlyConcatenate(ToMask, NewToMask) &&
14830 BitsProperlyConcatenate(FromMask, NewFromMask))
14831 return V;
14832 if (BitsProperlyConcatenate(NewToMask, ToMask) &&
14833 BitsProperlyConcatenate(NewFromMask, FromMask))
14834 return V;
14835
14836 return SDValue();
14837}
14838
14840 SDValue N0 = N->getOperand(0);
14841 SDValue N1 = N->getOperand(1);
14842
14843 if (N1.getOpcode() == ISD::AND) {
14844 // (bfi A, (and B, Mask1), Mask2) -> (bfi A, B, Mask2) iff
14845 // the bits being cleared by the AND are not demanded by the BFI.
14847 if (!N11C)
14848 return SDValue();
14849 unsigned InvMask = N->getConstantOperandVal(2);
14850 unsigned LSB = llvm::countr_zero(~InvMask);
14851 unsigned Width = llvm::bit_width<unsigned>(~InvMask) - LSB;
14852 assert(Width <
14853 static_cast<unsigned>(std::numeric_limits<unsigned>::digits) &&
14854 "undefined behavior");
14855 unsigned Mask = (1u << Width) - 1;
14856 unsigned Mask2 = N11C->getZExtValue();
14857 if ((Mask & (~Mask2)) == 0)
14858 return DAG.getNode(ARMISD::BFI, SDLoc(N), N->getValueType(0),
14859 N->getOperand(0), N1.getOperand(0), N->getOperand(2));
14860 return SDValue();
14861 }
14862
14863 // Look for another BFI to combine with.
14864 if (SDValue CombineBFI = FindBFIToCombineWith(N)) {
14865 // We've found a BFI.
14866 APInt ToMask1, FromMask1;
14867 SDValue From1 = ParseBFI(N, ToMask1, FromMask1);
14868
14869 APInt ToMask2, FromMask2;
14870 SDValue From2 = ParseBFI(CombineBFI.getNode(), ToMask2, FromMask2);
14871 assert(From1 == From2);
14872 (void)From2;
14873
14874 // Create a new BFI, combining the two together.
14875 APInt NewFromMask = FromMask1 | FromMask2;
14876 APInt NewToMask = ToMask1 | ToMask2;
14877
14878 EVT VT = N->getValueType(0);
14879 SDLoc dl(N);
14880
14881 if (NewFromMask[0] == 0)
14882 From1 = DAG.getNode(ISD::SRL, dl, VT, From1,
14883 DAG.getConstant(NewFromMask.countr_zero(), dl, VT));
14884 return DAG.getNode(ARMISD::BFI, dl, VT, CombineBFI.getOperand(0), From1,
14885 DAG.getConstant(~NewToMask, dl, VT));
14886 }
14887
14888 // Reassociate BFI(BFI (A, B, M1), C, M2) to BFI(BFI (A, C, M2), B, M1) so
14889 // that lower bit insertions are performed first, providing that M1 and M2
14890 // do no overlap. This can allow multiple BFI instructions to be combined
14891 // together by the other folds above.
14892 if (N->getOperand(0).getOpcode() == ARMISD::BFI) {
14893 APInt ToMask1 = ~N->getConstantOperandAPInt(2);
14894 APInt ToMask2 = ~N0.getConstantOperandAPInt(2);
14895
14896 if (!N0.hasOneUse() || (ToMask1 & ToMask2) != 0 ||
14897 ToMask1.countl_zero() < ToMask2.countl_zero())
14898 return SDValue();
14899
14900 EVT VT = N->getValueType(0);
14901 SDLoc dl(N);
14902 SDValue BFI1 = DAG.getNode(ARMISD::BFI, dl, VT, N0.getOperand(0),
14903 N->getOperand(1), N->getOperand(2));
14904 return DAG.getNode(ARMISD::BFI, dl, VT, BFI1, N0.getOperand(1),
14905 N0.getOperand(2));
14906 }
14907
14908 return SDValue();
14909}
14910
14911// Check that N is CMPZ(CSINC(0, 0, CC, X)),
14912// or CMPZ(CMOV(1, 0, CC, X))
14913// return X if valid.
14915 if (Cmp->getOpcode() != ARMISD::CMPZ || !isNullConstant(Cmp->getOperand(1)))
14916 return SDValue();
14917 SDValue CSInc = Cmp->getOperand(0);
14918
14919 // Ignore any `And 1` nodes that may not yet have been removed. We are
14920 // looking for a value that produces 1/0, so these have no effect on the
14921 // code.
14922 while (CSInc.getOpcode() == ISD::AND &&
14923 isa<ConstantSDNode>(CSInc.getOperand(1)) &&
14924 CSInc.getConstantOperandVal(1) == 1 && CSInc->hasOneUse())
14925 CSInc = CSInc.getOperand(0);
14926
14927 if (CSInc.getOpcode() == ARMISD::CSINC &&
14928 isNullConstant(CSInc.getOperand(0)) &&
14929 isNullConstant(CSInc.getOperand(1)) && CSInc->hasOneUse()) {
14931 return CSInc.getOperand(3);
14932 }
14933 if (CSInc.getOpcode() == ARMISD::CMOV && isOneConstant(CSInc.getOperand(0)) &&
14934 isNullConstant(CSInc.getOperand(1)) && CSInc->hasOneUse()) {
14936 return CSInc.getOperand(3);
14937 }
14938 if (CSInc.getOpcode() == ARMISD::CMOV && isOneConstant(CSInc.getOperand(1)) &&
14939 isNullConstant(CSInc.getOperand(0)) && CSInc->hasOneUse()) {
14942 return CSInc.getOperand(3);
14943 }
14944 return SDValue();
14945}
14946
14948 // Given CMPZ(CSINC(C, 0, 0, EQ), 0), we can just use C directly. As in
14949 // t92: flags = ARMISD::CMPZ t74, 0
14950 // t93: i32 = ARMISD::CSINC 0, 0, 1, t92
14951 // t96: flags = ARMISD::CMPZ t93, 0
14952 // t114: i32 = ARMISD::CSINV 0, 0, 0, t96
14954 if (SDValue C = IsCMPZCSINC(N, Cond))
14955 if (Cond == ARMCC::EQ)
14956 return C;
14957 return SDValue();
14958}
14959
14961 // Fold away an unnecessary CMPZ/CSINC
14962 // CSXYZ A, B, C1 (CMPZ (CSINC 0, 0, C2, D), 0) ->
14963 // if C1==EQ -> CSXYZ A, B, C2, D
14964 // if C1==NE -> CSXYZ A, B, NOT(C2), D
14966 if (SDValue C = IsCMPZCSINC(N->getOperand(3).getNode(), Cond)) {
14967 if (N->getConstantOperandVal(2) == ARMCC::EQ)
14968 return DAG.getNode(N->getOpcode(), SDLoc(N), MVT::i32, N->getOperand(0),
14969 N->getOperand(1),
14970 DAG.getConstant(Cond, SDLoc(N), MVT::i32), C);
14971 if (N->getConstantOperandVal(2) == ARMCC::NE)
14972 return DAG.getNode(
14973 N->getOpcode(), SDLoc(N), MVT::i32, N->getOperand(0),
14974 N->getOperand(1),
14976 }
14977 return SDValue();
14978}
14979
14980/// PerformVMOVRRDCombine - Target-specific dag combine xforms for
14981/// ARMISD::VMOVRRD.
14984 const ARMSubtarget *Subtarget) {
14985 // vmovrrd(vmovdrr x, y) -> x,y
14986 SDValue InDouble = N->getOperand(0);
14987 if (InDouble.getOpcode() == ARMISD::VMOVDRR && Subtarget->hasFP64())
14988 return DCI.CombineTo(N, InDouble.getOperand(0), InDouble.getOperand(1));
14989
14990 // vmovrrd(load f64) -> (load i32), (load i32)
14991 SDNode *InNode = InDouble.getNode();
14992 if (ISD::isNormalLoad(InNode) && InNode->hasOneUse() &&
14993 InNode->getValueType(0) == MVT::f64 &&
14994 InNode->getOperand(1).getOpcode() == ISD::FrameIndex &&
14995 !cast<LoadSDNode>(InNode)->isVolatile()) {
14996 // TODO: Should this be done for non-FrameIndex operands?
14997 LoadSDNode *LD = cast<LoadSDNode>(InNode);
14998
14999 SelectionDAG &DAG = DCI.DAG;
15000 SDLoc DL(LD);
15001 SDValue BasePtr = LD->getBasePtr();
15002 SDValue NewLD1 =
15003 DAG.getLoad(MVT::i32, DL, LD->getChain(), BasePtr, LD->getPointerInfo(),
15004 LD->getAlign(), LD->getMemOperand()->getFlags());
15005
15006 SDValue OffsetPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr,
15007 DAG.getConstant(4, DL, MVT::i32));
15008
15009 SDValue NewLD2 = DAG.getLoad(MVT::i32, DL, LD->getChain(), OffsetPtr,
15010 LD->getPointerInfo().getWithOffset(4),
15011 commonAlignment(LD->getAlign(), 4),
15012 LD->getMemOperand()->getFlags());
15013
15014 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), NewLD2.getValue(1));
15015 if (DCI.DAG.getDataLayout().isBigEndian())
15016 std::swap (NewLD1, NewLD2);
15017 SDValue Result = DCI.CombineTo(N, NewLD1, NewLD2);
15018 return Result;
15019 }
15020
15021 // VMOVRRD(extract(..(build_vector(a, b, c, d)))) -> a,b or c,d
15022 // VMOVRRD(extract(insert_vector(insert_vector(.., a, l1), b, l2))) -> a,b
15023 if (InDouble.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
15024 isa<ConstantSDNode>(InDouble.getOperand(1))) {
15025 SDValue BV = InDouble.getOperand(0);
15026 // Look up through any nop bitcasts and vector_reg_casts. bitcasts may
15027 // change lane order under big endian.
15028 bool BVSwap = BV.getOpcode() == ISD::BITCAST;
15029 while (
15030 (BV.getOpcode() == ISD::BITCAST ||
15031 BV.getOpcode() == ARMISD::VECTOR_REG_CAST) &&
15032 (BV.getValueType() == MVT::v2f64 || BV.getValueType() == MVT::v2i64)) {
15033 BVSwap = BV.getOpcode() == ISD::BITCAST;
15034 BV = BV.getOperand(0);
15035 }
15036 if (BV.getValueType() != MVT::v4i32)
15037 return SDValue();
15038
15039 // Handle buildvectors, pulling out the correct lane depending on
15040 // endianness.
15041 unsigned Offset = InDouble.getConstantOperandVal(1) == 1 ? 2 : 0;
15042 if (BV.getOpcode() == ISD::BUILD_VECTOR) {
15043 SDValue Op0 = BV.getOperand(Offset);
15044 SDValue Op1 = BV.getOperand(Offset + 1);
15045 if (!Subtarget->isLittle() && BVSwap)
15046 std::swap(Op0, Op1);
15047
15048 return DCI.DAG.getMergeValues({Op0, Op1}, SDLoc(N));
15049 }
15050
15051 // A chain of insert_vectors, grabbing the correct value of the chain of
15052 // inserts.
15053 SDValue Op0, Op1;
15054 while (BV.getOpcode() == ISD::INSERT_VECTOR_ELT) {
15055 if (isa<ConstantSDNode>(BV.getOperand(2))) {
15056 if (BV.getConstantOperandVal(2) == Offset && !Op0)
15057 Op0 = BV.getOperand(1);
15058 if (BV.getConstantOperandVal(2) == Offset + 1 && !Op1)
15059 Op1 = BV.getOperand(1);
15060 }
15061 BV = BV.getOperand(0);
15062 }
15063 if (!Subtarget->isLittle() && BVSwap)
15064 std::swap(Op0, Op1);
15065 if (Op0 && Op1)
15066 return DCI.DAG.getMergeValues({Op0, Op1}, SDLoc(N));
15067 }
15068
15069 return SDValue();
15070}
15071
15072/// PerformVMOVDRRCombine - Target-specific dag combine xforms for
15073/// ARMISD::VMOVDRR. This is also used for BUILD_VECTORs with 2 operands.
15075 // N=vmovrrd(X); vmovdrr(N:0, N:1) -> bit_convert(X)
15076 SDValue Op0 = N->getOperand(0);
15077 SDValue Op1 = N->getOperand(1);
15078 if (Op0.getOpcode() == ISD::BITCAST)
15079 Op0 = Op0.getOperand(0);
15080 if (Op1.getOpcode() == ISD::BITCAST)
15081 Op1 = Op1.getOperand(0);
15082 if (Op0.getOpcode() == ARMISD::VMOVRRD &&
15083 Op0.getNode() == Op1.getNode() &&
15084 Op0.getResNo() == 0 && Op1.getResNo() == 1)
15085 return DAG.getNode(ISD::BITCAST, SDLoc(N),
15086 N->getValueType(0), Op0.getOperand(0));
15087 return SDValue();
15088}
15089
15092 SDValue Op0 = N->getOperand(0);
15093
15094 // VMOVhr (VMOVrh (X)) -> X
15095 if (Op0->getOpcode() == ARMISD::VMOVrh)
15096 return Op0->getOperand(0);
15097
15098 // FullFP16: half values are passed in S-registers, and we don't
15099 // need any of the bitcast and moves:
15100 //
15101 // t2: f32,ch1,gl1? = CopyFromReg ch, Register:f32 %0, gl?
15102 // t5: i32 = bitcast t2
15103 // t18: f16 = ARMISD::VMOVhr t5
15104 // =>
15105 // tN: f16,ch2,gl2? = CopyFromReg ch, Register::f32 %0, gl?
15106 if (Op0->getOpcode() == ISD::BITCAST) {
15107 SDValue Copy = Op0->getOperand(0);
15108 if (Copy.getValueType() == MVT::f32 &&
15109 Copy->getOpcode() == ISD::CopyFromReg) {
15110 bool HasGlue = Copy->getNumOperands() == 3;
15111 SDValue Ops[] = {Copy->getOperand(0), Copy->getOperand(1),
15112 HasGlue ? Copy->getOperand(2) : SDValue()};
15113 EVT OutTys[] = {N->getValueType(0), MVT::Other, MVT::Glue};
15114 SDValue NewCopy =
15116 DCI.DAG.getVTList(ArrayRef(OutTys, HasGlue ? 3 : 2)),
15117 ArrayRef(Ops, HasGlue ? 3 : 2));
15118
15119 // Update Users, Chains, and Potential Glue.
15120 DCI.DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), NewCopy.getValue(0));
15121 DCI.DAG.ReplaceAllUsesOfValueWith(Copy.getValue(1), NewCopy.getValue(1));
15122 if (HasGlue)
15123 DCI.DAG.ReplaceAllUsesOfValueWith(Copy.getValue(2),
15124 NewCopy.getValue(2));
15125
15126 return NewCopy;
15127 }
15128 }
15129
15130 // fold (VMOVhr (load x)) -> (load (f16*)x)
15131 if (LoadSDNode *LN0 = dyn_cast<LoadSDNode>(Op0)) {
15132 if (LN0->hasOneUse() && LN0->isUnindexed() &&
15133 LN0->getMemoryVT() == MVT::i16) {
15134 SDValue Load =
15135 DCI.DAG.getLoad(N->getValueType(0), SDLoc(N), LN0->getChain(),
15136 LN0->getBasePtr(), LN0->getMemOperand());
15137 DCI.DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Load.getValue(0));
15138 DCI.DAG.ReplaceAllUsesOfValueWith(Op0.getValue(1), Load.getValue(1));
15139 return Load;
15140 }
15141 }
15142
15143 // Only the bottom 16 bits of the source register are used.
15144 APInt DemandedMask = APInt::getLowBitsSet(32, 16);
15145 const TargetLowering &TLI = DCI.DAG.getTargetLoweringInfo();
15146 if (TLI.SimplifyDemandedBits(Op0, DemandedMask, DCI))
15147 return SDValue(N, 0);
15148
15149 return SDValue();
15150}
15151
15153 SDValue N0 = N->getOperand(0);
15154 EVT VT = N->getValueType(0);
15155
15156 // fold (VMOVrh (fpconst x)) -> const x
15158 APFloat V = C->getValueAPF();
15159 return DAG.getConstant(V.bitcastToAPInt().getZExtValue(), SDLoc(N), VT);
15160 }
15161
15162 // fold (VMOVrh (load x)) -> (zextload (i16*)x)
15163 if (ISD::isNormalLoad(N0.getNode()) && N0.hasOneUse()) {
15164 LoadSDNode *LN0 = cast<LoadSDNode>(N0);
15165
15166 SDValue Load =
15167 DAG.getExtLoad(ISD::ZEXTLOAD, SDLoc(N), VT, LN0->getChain(),
15168 LN0->getBasePtr(), MVT::i16, LN0->getMemOperand());
15169 DAG.ReplaceAllUsesOfValueWith(SDValue(N, 0), Load.getValue(0));
15170 DAG.ReplaceAllUsesOfValueWith(N0.getValue(1), Load.getValue(1));
15171 return Load;
15172 }
15173
15174 // Fold VMOVrh(extract(x, n)) -> vgetlaneu(x, n)
15175 if (N0->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
15177 return DAG.getNode(ARMISD::VGETLANEu, SDLoc(N), VT, N0->getOperand(0),
15178 N0->getOperand(1));
15179
15180 return SDValue();
15181}
15182
15183/// hasNormalLoadOperand - Check if any of the operands of a BUILD_VECTOR node
15184/// are normal, non-volatile loads. If so, it is profitable to bitcast an
15185/// i64 vector to have f64 elements, since the value can then be loaded
15186/// directly into a VFP register.
15188 unsigned NumElts = N->getValueType(0).getVectorNumElements();
15189 for (unsigned i = 0; i < NumElts; ++i) {
15190 SDNode *Elt = N->getOperand(i).getNode();
15191 if (ISD::isNormalLoad(Elt) && !cast<LoadSDNode>(Elt)->isVolatile())
15192 return true;
15193 }
15194 return false;
15195}
15196
15197/// PerformBUILD_VECTORCombine - Target-specific dag combine xforms for
15198/// ISD::BUILD_VECTOR.
15201 const ARMSubtarget *Subtarget) {
15202 // build_vector(N=ARMISD::VMOVRRD(X), N:1) -> bit_convert(X):
15203 // VMOVRRD is introduced when legalizing i64 types. It forces the i64 value
15204 // into a pair of GPRs, which is fine when the value is used as a scalar,
15205 // but if the i64 value is converted to a vector, we need to undo the VMOVRRD.
15206 SelectionDAG &DAG = DCI.DAG;
15207 if (N->getNumOperands() == 2)
15208 if (SDValue RV = PerformVMOVDRRCombine(N, DAG))
15209 return RV;
15210
15211 // Load i64 elements as f64 values so that type legalization does not split
15212 // them up into i32 values.
15213 EVT VT = N->getValueType(0);
15214 if (VT.getVectorElementType() != MVT::i64 || !hasNormalLoadOperand(N))
15215 return SDValue();
15216 SDLoc dl(N);
15218 unsigned NumElts = VT.getVectorNumElements();
15219 for (unsigned i = 0; i < NumElts; ++i) {
15220 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::f64, N->getOperand(i));
15221 Ops.push_back(V);
15222 // Make the DAGCombiner fold the bitcast.
15223 DCI.AddToWorklist(V.getNode());
15224 }
15225 EVT FloatVT = EVT::getVectorVT(*DAG.getContext(), MVT::f64, NumElts);
15226 SDValue BV = DAG.getBuildVector(FloatVT, dl, Ops);
15227 return DAG.getNode(ISD::BITCAST, dl, VT, BV);
15228}
15229
15230/// Target-specific dag combine xforms for ARMISD::BUILD_VECTOR.
15231static SDValue
15233 // ARMISD::BUILD_VECTOR is introduced when legalizing ISD::BUILD_VECTOR.
15234 // At that time, we may have inserted bitcasts from integer to float.
15235 // If these bitcasts have survived DAGCombine, change the lowering of this
15236 // BUILD_VECTOR in something more vector friendly, i.e., that does not
15237 // force to use floating point types.
15238
15239 // Make sure we can change the type of the vector.
15240 // This is possible iff:
15241 // 1. The vector is only used in a bitcast to a integer type. I.e.,
15242 // 1.1. Vector is used only once.
15243 // 1.2. Use is a bit convert to an integer type.
15244 // 2. The size of its operands are 32-bits (64-bits are not legal).
15245 EVT VT = N->getValueType(0);
15246 EVT EltVT = VT.getVectorElementType();
15247
15248 // Check 1.1. and 2.
15249 if (EltVT.getSizeInBits() != 32 || !N->hasOneUse())
15250 return SDValue();
15251
15252 // By construction, the input type must be float.
15253 assert(EltVT == MVT::f32 && "Unexpected type!");
15254
15255 // Check 1.2.
15256 SDNode *Use = *N->user_begin();
15257 if (Use->getOpcode() != ISD::BITCAST ||
15258 Use->getValueType(0).isFloatingPoint())
15259 return SDValue();
15260
15261 // Check profitability.
15262 // Model is, if more than half of the relevant operands are bitcast from
15263 // i32, turn the build_vector into a sequence of insert_vector_elt.
15264 // Relevant operands are everything that is not statically
15265 // (i.e., at compile time) bitcasted.
15266 unsigned NumOfBitCastedElts = 0;
15267 unsigned NumElts = VT.getVectorNumElements();
15268 unsigned NumOfRelevantElts = NumElts;
15269 for (unsigned Idx = 0; Idx < NumElts; ++Idx) {
15270 SDValue Elt = N->getOperand(Idx);
15271 if (Elt->getOpcode() == ISD::BITCAST) {
15272 // Assume only bit cast to i32 will go away.
15273 if (Elt->getOperand(0).getValueType() == MVT::i32)
15274 ++NumOfBitCastedElts;
15275 } else if (Elt.isUndef() || isa<ConstantSDNode>(Elt))
15276 // Constants are statically casted, thus do not count them as
15277 // relevant operands.
15278 --NumOfRelevantElts;
15279 }
15280
15281 // Check if more than half of the elements require a non-free bitcast.
15282 if (NumOfBitCastedElts <= NumOfRelevantElts / 2)
15283 return SDValue();
15284
15285 SelectionDAG &DAG = DCI.DAG;
15286 // Create the new vector type.
15287 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElts);
15288 // Check if the type is legal.
15289 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
15290 if (!TLI.isTypeLegal(VecVT))
15291 return SDValue();
15292
15293 // Combine:
15294 // ARMISD::BUILD_VECTOR E1, E2, ..., EN.
15295 // => BITCAST INSERT_VECTOR_ELT
15296 // (INSERT_VECTOR_ELT (...), (BITCAST EN-1), N-1),
15297 // (BITCAST EN), N.
15298 SDValue Vec = DAG.getUNDEF(VecVT);
15299 SDLoc dl(N);
15300 for (unsigned Idx = 0 ; Idx < NumElts; ++Idx) {
15301 SDValue V = N->getOperand(Idx);
15302 if (V.isUndef())
15303 continue;
15304 if (V.getOpcode() == ISD::BITCAST &&
15305 V->getOperand(0).getValueType() == MVT::i32)
15306 // Fold obvious case.
15307 V = V.getOperand(0);
15308 else {
15309 V = DAG.getNode(ISD::BITCAST, SDLoc(V), MVT::i32, V);
15310 // Make the DAGCombiner fold the bitcasts.
15311 DCI.AddToWorklist(V.getNode());
15312 }
15313 SDValue LaneIdx = DAG.getConstant(Idx, dl, MVT::i32);
15314 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, VecVT, Vec, V, LaneIdx);
15315 }
15316 Vec = DAG.getNode(ISD::BITCAST, dl, VT, Vec);
15317 // Make the DAGCombiner fold the bitcasts.
15318 DCI.AddToWorklist(Vec.getNode());
15319 return Vec;
15320}
15321
15322static SDValue
15324 EVT VT = N->getValueType(0);
15325 SDValue Op = N->getOperand(0);
15326 SDLoc dl(N);
15327
15328 // PREDICATE_CAST(PREDICATE_CAST(x)) == PREDICATE_CAST(x)
15329 if (Op->getOpcode() == ARMISD::PREDICATE_CAST) {
15330 // If the valuetypes are the same, we can remove the cast entirely.
15331 if (Op->getOperand(0).getValueType() == VT)
15332 return Op->getOperand(0);
15333 return DCI.DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT, Op->getOperand(0));
15334 }
15335
15336 // Turn pred_cast(xor x, -1) into xor(pred_cast x, -1), in order to produce
15337 // more VPNOT which might get folded as else predicates.
15338 if (Op.getValueType() == MVT::i32 && isBitwiseNot(Op)) {
15339 SDValue X =
15340 DCI.DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT, Op->getOperand(0));
15341 SDValue C = DCI.DAG.getNode(ARMISD::PREDICATE_CAST, dl, VT,
15342 DCI.DAG.getConstant(65535, dl, MVT::i32));
15343 return DCI.DAG.getNode(ISD::XOR, dl, VT, X, C);
15344 }
15345
15346 // Only the bottom 16 bits of the source register are used.
15347 if (Op.getValueType() == MVT::i32) {
15348 APInt DemandedMask = APInt::getLowBitsSet(32, 16);
15349 const TargetLowering &TLI = DCI.DAG.getTargetLoweringInfo();
15350 if (TLI.SimplifyDemandedBits(Op, DemandedMask, DCI))
15351 return SDValue(N, 0);
15352 }
15353 return SDValue();
15354}
15355
15357 const ARMSubtarget *ST) {
15358 EVT VT = N->getValueType(0);
15359 SDValue Op = N->getOperand(0);
15360 SDLoc dl(N);
15361
15362 // Under Little endian, a VECTOR_REG_CAST is equivalent to a BITCAST
15363 if (ST->isLittle())
15364 return DAG.getNode(ISD::BITCAST, dl, VT, Op);
15365
15366 // VT VECTOR_REG_CAST (VT Op) -> Op
15367 if (Op.getValueType() == VT)
15368 return Op;
15369 // VECTOR_REG_CAST undef -> undef
15370 if (Op.isUndef())
15371 return DAG.getUNDEF(VT);
15372
15373 // VECTOR_REG_CAST(VECTOR_REG_CAST(x)) == VECTOR_REG_CAST(x)
15374 if (Op->getOpcode() == ARMISD::VECTOR_REG_CAST) {
15375 // If the valuetypes are the same, we can remove the cast entirely.
15376 if (Op->getOperand(0).getValueType() == VT)
15377 return Op->getOperand(0);
15378 return DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, VT, Op->getOperand(0));
15379 }
15380
15381 return SDValue();
15382}
15383
15385 const ARMSubtarget *Subtarget) {
15386 if (!Subtarget->hasMVEIntegerOps())
15387 return SDValue();
15388
15389 EVT VT = N->getValueType(0);
15390 SDValue Op0 = N->getOperand(0);
15391 SDValue Op1 = N->getOperand(1);
15392 ARMCC::CondCodes Cond = (ARMCC::CondCodes)N->getConstantOperandVal(2);
15393 SDLoc dl(N);
15394
15395 // vcmp X, 0, cc -> vcmpz X, cc
15396 if (isZeroVector(Op1))
15397 return DAG.getNode(ARMISD::VCMPZ, dl, VT, Op0, N->getOperand(2));
15398
15399 unsigned SwappedCond = getSwappedCondition(Cond);
15400 if (isValidMVECond(SwappedCond, VT.isFloatingPoint())) {
15401 // vcmp 0, X, cc -> vcmpz X, reversed(cc)
15402 if (isZeroVector(Op0))
15403 return DAG.getNode(ARMISD::VCMPZ, dl, VT, Op1,
15404 DAG.getConstant(SwappedCond, dl, MVT::i32));
15405 // vcmp vdup(Y), X, cc -> vcmp X, vdup(Y), reversed(cc)
15406 if (Op0->getOpcode() == ARMISD::VDUP && Op1->getOpcode() != ARMISD::VDUP)
15407 return DAG.getNode(ARMISD::VCMP, dl, VT, Op1, Op0,
15408 DAG.getConstant(SwappedCond, dl, MVT::i32));
15409 }
15410
15411 return SDValue();
15412}
15413
15414/// PerformInsertEltCombine - Target-specific dag combine xforms for
15415/// ISD::INSERT_VECTOR_ELT.
15418 // Bitcast an i64 load inserted into a vector to f64.
15419 // Otherwise, the i64 value will be legalized to a pair of i32 values.
15420 EVT VT = N->getValueType(0);
15421 SDNode *Elt = N->getOperand(1).getNode();
15422 if (VT.getVectorElementType() != MVT::i64 ||
15423 !ISD::isNormalLoad(Elt) || cast<LoadSDNode>(Elt)->isVolatile())
15424 return SDValue();
15425
15426 SelectionDAG &DAG = DCI.DAG;
15427 SDLoc dl(N);
15428 EVT FloatVT = EVT::getVectorVT(*DAG.getContext(), MVT::f64,
15430 SDValue Vec = DAG.getNode(ISD::BITCAST, dl, FloatVT, N->getOperand(0));
15431 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::f64, N->getOperand(1));
15432 // Make the DAGCombiner fold the bitcasts.
15433 DCI.AddToWorklist(Vec.getNode());
15434 DCI.AddToWorklist(V.getNode());
15435 SDValue InsElt = DAG.getNode(ISD::INSERT_VECTOR_ELT, dl, FloatVT,
15436 Vec, V, N->getOperand(2));
15437 return DAG.getNode(ISD::BITCAST, dl, VT, InsElt);
15438}
15439
15440// Convert a pair of extracts from the same base vector to a VMOVRRD. Either
15441// directly or bitcast to an integer if the original is a float vector.
15442// extract(x, n); extract(x, n+1) -> VMOVRRD(extract v2f64 x, n/2)
15443// bitcast(extract(x, n)); bitcast(extract(x, n+1)) -> VMOVRRD(extract x, n/2)
15444static SDValue
15446 EVT VT = N->getValueType(0);
15447 SDLoc dl(N);
15448
15449 if (!DCI.isAfterLegalizeDAG() || VT != MVT::i32 ||
15450 !DCI.DAG.getTargetLoweringInfo().isTypeLegal(MVT::f64))
15451 return SDValue();
15452
15453 SDValue Ext = SDValue(N, 0);
15454 if (Ext.getOpcode() == ISD::BITCAST &&
15455 Ext.getOperand(0).getValueType() == MVT::f32)
15456 Ext = Ext.getOperand(0);
15457 if (Ext.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
15459 Ext.getConstantOperandVal(1) % 2 != 0)
15460 return SDValue();
15461 if (Ext->hasOneUse() && (Ext->user_begin()->getOpcode() == ISD::SINT_TO_FP ||
15462 Ext->user_begin()->getOpcode() == ISD::UINT_TO_FP))
15463 return SDValue();
15464
15465 SDValue Op0 = Ext.getOperand(0);
15466 EVT VecVT = Op0.getValueType();
15467 unsigned ResNo = Op0.getResNo();
15468 unsigned Lane = Ext.getConstantOperandVal(1);
15469 if (VecVT.getVectorNumElements() != 4)
15470 return SDValue();
15471
15472 // Find another extract, of Lane + 1
15473 auto OtherIt = find_if(Op0->users(), [&](SDNode *V) {
15474 return V->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
15475 isa<ConstantSDNode>(V->getOperand(1)) &&
15476 V->getConstantOperandVal(1) == Lane + 1 &&
15477 V->getOperand(0).getResNo() == ResNo;
15478 });
15479 if (OtherIt == Op0->users().end())
15480 return SDValue();
15481
15482 // For float extracts, we need to be converting to a i32 for both vector
15483 // lanes.
15484 SDValue OtherExt(*OtherIt, 0);
15485 if (OtherExt.getValueType() != MVT::i32) {
15486 if (!OtherExt->hasOneUse() ||
15487 OtherExt->user_begin()->getOpcode() != ISD::BITCAST ||
15488 OtherExt->user_begin()->getValueType(0) != MVT::i32)
15489 return SDValue();
15490 OtherExt = SDValue(*OtherExt->user_begin(), 0);
15491 }
15492
15493 // Convert the type to a f64 and extract with a VMOVRRD.
15494 SDValue F64 = DCI.DAG.getNode(
15495 ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64,
15496 DCI.DAG.getNode(ARMISD::VECTOR_REG_CAST, dl, MVT::v2f64, Op0),
15497 DCI.DAG.getConstant(Ext.getConstantOperandVal(1) / 2, dl, MVT::i32));
15498 SDValue VMOVRRD =
15499 DCI.DAG.getNode(ARMISD::VMOVRRD, dl, {MVT::i32, MVT::i32}, F64);
15500
15501 DCI.CombineTo(OtherExt.getNode(), SDValue(VMOVRRD.getNode(), 1));
15502 return VMOVRRD;
15503}
15504
15507 const ARMSubtarget *ST) {
15508 SDValue Op0 = N->getOperand(0);
15509 EVT VT = N->getValueType(0);
15510 SDLoc dl(N);
15511
15512 // extract (vdup x) -> x
15513 if (Op0->getOpcode() == ARMISD::VDUP) {
15514 SDValue X = Op0->getOperand(0);
15515 if (VT == MVT::f16 && X.getValueType() == MVT::i32)
15516 return DCI.DAG.getNode(ARMISD::VMOVhr, dl, VT, X);
15517 if (VT == MVT::i32 && X.getValueType() == MVT::f16)
15518 return DCI.DAG.getNode(ARMISD::VMOVrh, dl, VT, X);
15519 if (VT == MVT::f32 && X.getValueType() == MVT::i32)
15520 return DCI.DAG.getNode(ISD::BITCAST, dl, VT, X);
15521
15522 while (X.getValueType() != VT && X->getOpcode() == ISD::BITCAST)
15523 X = X->getOperand(0);
15524 if (X.getValueType() == VT)
15525 return X;
15526 }
15527
15528 // extract ARM_BUILD_VECTOR -> x
15529 if (Op0->getOpcode() == ARMISD::BUILD_VECTOR &&
15530 isa<ConstantSDNode>(N->getOperand(1)) &&
15531 N->getConstantOperandVal(1) < Op0.getNumOperands()) {
15532 return Op0.getOperand(N->getConstantOperandVal(1));
15533 }
15534
15535 // extract(bitcast(BUILD_VECTOR(VMOVDRR(a, b), ..))) -> a or b
15536 if (Op0.getValueType() == MVT::v4i32 &&
15537 isa<ConstantSDNode>(N->getOperand(1)) &&
15538 Op0.getOpcode() == ISD::BITCAST &&
15540 Op0.getOperand(0).getValueType() == MVT::v2f64) {
15541 SDValue BV = Op0.getOperand(0);
15542 unsigned Offset = N->getConstantOperandVal(1);
15543 SDValue MOV = BV.getOperand(Offset < 2 ? 0 : 1);
15544 if (MOV.getOpcode() == ARMISD::VMOVDRR)
15545 return MOV.getOperand(ST->isLittle() ? Offset % 2 : 1 - Offset % 2);
15546 }
15547
15548 // extract x, n; extract x, n+1 -> VMOVRRD x
15549 if (SDValue R = PerformExtractEltToVMOVRRD(N, DCI))
15550 return R;
15551
15552 // extract (MVETrunc(x)) -> extract x
15553 if (Op0->getOpcode() == ARMISD::MVETRUNC) {
15554 unsigned Idx = N->getConstantOperandVal(1);
15555 unsigned Vec =
15557 unsigned SubIdx =
15559 return DCI.DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, VT, Op0.getOperand(Vec),
15560 DCI.DAG.getConstant(SubIdx, dl, MVT::i32));
15561 }
15562
15563 // extract(bitcast(BUILD_VECTOR(extract(bitcast(a)), ..))) -> extract(a)
15564 if (ST->isLittle() && Op0.getOpcode() == ISD::BITCAST &&
15566 isa<ConstantSDNode>(N->getOperand(1)) &&
15569 unsigned Lane = N->getConstantOperandVal(1);
15570 EVT ExtVT = Op0.getValueType();
15571 EVT BVVT = Op0.getOperand(0).getValueType();
15572 unsigned BVLane =
15573 (Lane * BVVT.getVectorNumElements()) / ExtVT.getVectorNumElements();
15574 assert(BVLane < Op0.getOperand(0).getNumOperands());
15575 SDValue Ext = Op0.getOperand(0).getOperand(BVLane);
15576 if (Ext.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
15577 Ext.getOperand(0).getOpcode() == ISD::BITCAST &&
15579 Ext.getOperand(0).getOperand(0).getValueType() == ExtVT) {
15580 unsigned InnerLane = Ext.getConstantOperandVal(1);
15581 unsigned BVSubLane = Lane - (BVLane * ExtVT.getVectorNumElements()) /
15582 BVVT.getVectorNumElements();
15583 unsigned FinalLane = (InnerLane * ExtVT.getVectorNumElements()) /
15584 BVVT.getVectorNumElements() +
15585 BVSubLane;
15586 return DCI.DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, VT,
15587 Ext.getOperand(0).getOperand(0),
15588 DCI.DAG.getConstant(FinalLane, dl, MVT::i32));
15589 }
15590 }
15591
15592 return SDValue();
15593}
15594
15596 SDValue Op = N->getOperand(0);
15597 EVT VT = N->getValueType(0);
15598
15599 // sext_inreg(VGETLANEu) -> VGETLANEs
15600 if (Op.getOpcode() == ARMISD::VGETLANEu &&
15601 cast<VTSDNode>(N->getOperand(1))->getVT() ==
15602 Op.getOperand(0).getValueType().getScalarType())
15603 return DAG.getNode(ARMISD::VGETLANEs, SDLoc(N), VT, Op.getOperand(0),
15604 Op.getOperand(1));
15605
15606 return SDValue();
15607}
15608
15609static SDValue
15611 SDValue Vec = N->getOperand(0);
15612 SDValue SubVec = N->getOperand(1);
15613 uint64_t IdxVal = N->getConstantOperandVal(2);
15614 EVT VecVT = Vec.getValueType();
15615 EVT SubVT = SubVec.getValueType();
15616
15617 // Only do this for legal fixed vector types.
15618 if (!VecVT.isFixedLengthVector() ||
15619 !DCI.DAG.getTargetLoweringInfo().isTypeLegal(VecVT) ||
15621 return SDValue();
15622
15623 // Ignore widening patterns.
15624 if (IdxVal == 0 && Vec.isUndef())
15625 return SDValue();
15626
15627 // Subvector must be half the width and an "aligned" insertion.
15628 unsigned NumSubElts = SubVT.getVectorNumElements();
15629 if ((SubVT.getSizeInBits() * 2) != VecVT.getSizeInBits() ||
15630 (IdxVal != 0 && IdxVal != NumSubElts))
15631 return SDValue();
15632
15633 // Fold insert_subvector -> concat_vectors
15634 // insert_subvector(Vec,Sub,lo) -> concat_vectors(Sub,extract(Vec,hi))
15635 // insert_subvector(Vec,Sub,hi) -> concat_vectors(extract(Vec,lo),Sub)
15636 SDLoc DL(N);
15637 SDValue Lo, Hi;
15638 if (IdxVal == 0) {
15639 Lo = SubVec;
15640 Hi = DCI.DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SubVT, Vec,
15641 DCI.DAG.getVectorIdxConstant(NumSubElts, DL));
15642 } else {
15643 Lo = DCI.DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, SubVT, Vec,
15644 DCI.DAG.getVectorIdxConstant(0, DL));
15645 Hi = SubVec;
15646 }
15647 return DCI.DAG.getNode(ISD::CONCAT_VECTORS, DL, VecVT, Lo, Hi);
15648}
15649
15650// shuffle(MVETrunc(x, y)) -> VMOVN(x, y)
15652 SelectionDAG &DAG) {
15653 SDValue Trunc = N->getOperand(0);
15654 EVT VT = Trunc.getValueType();
15655 if (Trunc.getOpcode() != ARMISD::MVETRUNC || !N->getOperand(1).isUndef())
15656 return SDValue();
15657
15658 SDLoc DL(Trunc);
15659 if (isVMOVNTruncMask(N->getMask(), VT, false))
15660 return DAG.getNode(
15661 ARMISD::VMOVN, DL, VT,
15662 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, Trunc.getOperand(0)),
15663 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, Trunc.getOperand(1)),
15664 DAG.getConstant(1, DL, MVT::i32));
15665 else if (isVMOVNTruncMask(N->getMask(), VT, true))
15666 return DAG.getNode(
15667 ARMISD::VMOVN, DL, VT,
15668 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, Trunc.getOperand(1)),
15669 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, Trunc.getOperand(0)),
15670 DAG.getConstant(1, DL, MVT::i32));
15671 return SDValue();
15672}
15673
15674/// PerformVECTOR_SHUFFLECombine - Target-specific dag combine xforms for
15675/// ISD::VECTOR_SHUFFLE.
15678 return R;
15679
15680 // The LLVM shufflevector instruction does not require the shuffle mask
15681 // length to match the operand vector length, but ISD::VECTOR_SHUFFLE does
15682 // have that requirement. When translating to ISD::VECTOR_SHUFFLE, if the
15683 // operands do not match the mask length, they are extended by concatenating
15684 // them with undef vectors. That is probably the right thing for other
15685 // targets, but for NEON it is better to concatenate two double-register
15686 // size vector operands into a single quad-register size vector. Do that
15687 // transformation here:
15688 // shuffle(concat(v1, undef), concat(v2, undef)) ->
15689 // shuffle(concat(v1, v2), undef)
15690 SDValue Op0 = N->getOperand(0);
15691 SDValue Op1 = N->getOperand(1);
15692 if (Op0.getOpcode() != ISD::CONCAT_VECTORS ||
15693 Op1.getOpcode() != ISD::CONCAT_VECTORS ||
15694 Op0.getNumOperands() != 2 ||
15695 Op1.getNumOperands() != 2)
15696 return SDValue();
15697 SDValue Concat0Op1 = Op0.getOperand(1);
15698 SDValue Concat1Op1 = Op1.getOperand(1);
15699 if (!Concat0Op1.isUndef() || !Concat1Op1.isUndef())
15700 return SDValue();
15701 // Skip the transformation if any of the types are illegal.
15702 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
15703 EVT VT = N->getValueType(0);
15704 if (!TLI.isTypeLegal(VT) ||
15705 !TLI.isTypeLegal(Concat0Op1.getValueType()) ||
15706 !TLI.isTypeLegal(Concat1Op1.getValueType()))
15707 return SDValue();
15708
15709 SDValue NewConcat = DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(N), VT,
15710 Op0.getOperand(0), Op1.getOperand(0));
15711 // Translate the shuffle mask.
15712 SmallVector<int, 16> NewMask;
15713 unsigned NumElts = VT.getVectorNumElements();
15714 unsigned HalfElts = NumElts/2;
15716 for (unsigned n = 0; n < NumElts; ++n) {
15717 int MaskElt = SVN->getMaskElt(n);
15718 int NewElt = -1;
15719 if (MaskElt < (int)HalfElts)
15720 NewElt = MaskElt;
15721 else if (MaskElt >= (int)NumElts && MaskElt < (int)(NumElts + HalfElts))
15722 NewElt = HalfElts + MaskElt - NumElts;
15723 NewMask.push_back(NewElt);
15724 }
15725 return DAG.getVectorShuffle(VT, SDLoc(N), NewConcat,
15726 DAG.getUNDEF(VT), NewMask);
15727}
15728
15729/// Load/store instruction that can be merged with a base address
15730/// update
15735 unsigned AddrOpIdx;
15736};
15737
15739 /// Instruction that updates a pointer
15741 /// Pointer increment operand
15743 /// Pointer increment value if it is a constant, or 0 otherwise
15744 unsigned ConstInc;
15745};
15746
15748 // Check that the add is independent of the load/store.
15749 // Otherwise, folding it would create a cycle. Search through Addr
15750 // as well, since the User may not be a direct user of Addr and
15751 // only share a base pointer.
15754 Worklist.push_back(N);
15755 Worklist.push_back(User);
15756 const unsigned MaxSteps = 1024;
15757 if (SDNode::hasPredecessorHelper(N, Visited, Worklist, MaxSteps) ||
15758 SDNode::hasPredecessorHelper(User, Visited, Worklist, MaxSteps))
15759 return false;
15760 return true;
15761}
15762
15764 struct BaseUpdateUser &User,
15765 bool SimpleConstIncOnly,
15767 SelectionDAG &DAG = DCI.DAG;
15768 SDNode *N = Target.N;
15769 MemSDNode *MemN = cast<MemSDNode>(N);
15770 SDLoc dl(N);
15771
15772 // Find the new opcode for the updating load/store.
15773 bool isLoadOp = true;
15774 bool isLaneOp = false;
15775 // Workaround for vst1x and vld1x intrinsics which do not have alignment
15776 // as an operand.
15777 bool hasAlignment = true;
15778 unsigned NewOpc = 0;
15779 unsigned NumVecs = 0;
15780 if (Target.isIntrinsic) {
15781 unsigned IntNo = N->getConstantOperandVal(1);
15782 switch (IntNo) {
15783 default:
15784 llvm_unreachable("unexpected intrinsic for Neon base update");
15785 case Intrinsic::arm_neon_vld1:
15786 NewOpc = ARMISD::VLD1_UPD;
15787 NumVecs = 1;
15788 break;
15789 case Intrinsic::arm_neon_vld2:
15790 NewOpc = ARMISD::VLD2_UPD;
15791 NumVecs = 2;
15792 break;
15793 case Intrinsic::arm_neon_vld3:
15794 NewOpc = ARMISD::VLD3_UPD;
15795 NumVecs = 3;
15796 break;
15797 case Intrinsic::arm_neon_vld4:
15798 NewOpc = ARMISD::VLD4_UPD;
15799 NumVecs = 4;
15800 break;
15801 case Intrinsic::arm_neon_vld1x2:
15802 NewOpc = ARMISD::VLD1x2_UPD;
15803 NumVecs = 2;
15804 hasAlignment = false;
15805 break;
15806 case Intrinsic::arm_neon_vld1x3:
15807 NewOpc = ARMISD::VLD1x3_UPD;
15808 NumVecs = 3;
15809 hasAlignment = false;
15810 break;
15811 case Intrinsic::arm_neon_vld1x4:
15812 NewOpc = ARMISD::VLD1x4_UPD;
15813 NumVecs = 4;
15814 hasAlignment = false;
15815 break;
15816 case Intrinsic::arm_neon_vld2dup:
15817 NewOpc = ARMISD::VLD2DUP_UPD;
15818 NumVecs = 2;
15819 break;
15820 case Intrinsic::arm_neon_vld3dup:
15821 NewOpc = ARMISD::VLD3DUP_UPD;
15822 NumVecs = 3;
15823 break;
15824 case Intrinsic::arm_neon_vld4dup:
15825 NewOpc = ARMISD::VLD4DUP_UPD;
15826 NumVecs = 4;
15827 break;
15828 case Intrinsic::arm_neon_vld2lane:
15829 NewOpc = ARMISD::VLD2LN_UPD;
15830 NumVecs = 2;
15831 isLaneOp = true;
15832 break;
15833 case Intrinsic::arm_neon_vld3lane:
15834 NewOpc = ARMISD::VLD3LN_UPD;
15835 NumVecs = 3;
15836 isLaneOp = true;
15837 break;
15838 case Intrinsic::arm_neon_vld4lane:
15839 NewOpc = ARMISD::VLD4LN_UPD;
15840 NumVecs = 4;
15841 isLaneOp = true;
15842 break;
15843 case Intrinsic::arm_neon_vst1:
15844 NewOpc = ARMISD::VST1_UPD;
15845 NumVecs = 1;
15846 isLoadOp = false;
15847 break;
15848 case Intrinsic::arm_neon_vst2:
15849 NewOpc = ARMISD::VST2_UPD;
15850 NumVecs = 2;
15851 isLoadOp = false;
15852 break;
15853 case Intrinsic::arm_neon_vst3:
15854 NewOpc = ARMISD::VST3_UPD;
15855 NumVecs = 3;
15856 isLoadOp = false;
15857 break;
15858 case Intrinsic::arm_neon_vst4:
15859 NewOpc = ARMISD::VST4_UPD;
15860 NumVecs = 4;
15861 isLoadOp = false;
15862 break;
15863 case Intrinsic::arm_neon_vst2lane:
15864 NewOpc = ARMISD::VST2LN_UPD;
15865 NumVecs = 2;
15866 isLoadOp = false;
15867 isLaneOp = true;
15868 break;
15869 case Intrinsic::arm_neon_vst3lane:
15870 NewOpc = ARMISD::VST3LN_UPD;
15871 NumVecs = 3;
15872 isLoadOp = false;
15873 isLaneOp = true;
15874 break;
15875 case Intrinsic::arm_neon_vst4lane:
15876 NewOpc = ARMISD::VST4LN_UPD;
15877 NumVecs = 4;
15878 isLoadOp = false;
15879 isLaneOp = true;
15880 break;
15881 case Intrinsic::arm_neon_vst1x2:
15882 NewOpc = ARMISD::VST1x2_UPD;
15883 NumVecs = 2;
15884 isLoadOp = false;
15885 hasAlignment = false;
15886 break;
15887 case Intrinsic::arm_neon_vst1x3:
15888 NewOpc = ARMISD::VST1x3_UPD;
15889 NumVecs = 3;
15890 isLoadOp = false;
15891 hasAlignment = false;
15892 break;
15893 case Intrinsic::arm_neon_vst1x4:
15894 NewOpc = ARMISD::VST1x4_UPD;
15895 NumVecs = 4;
15896 isLoadOp = false;
15897 hasAlignment = false;
15898 break;
15899 }
15900 } else {
15901 isLaneOp = true;
15902 switch (N->getOpcode()) {
15903 default:
15904 llvm_unreachable("unexpected opcode for Neon base update");
15905 case ARMISD::VLD1DUP:
15906 NewOpc = ARMISD::VLD1DUP_UPD;
15907 NumVecs = 1;
15908 break;
15909 case ARMISD::VLD2DUP:
15910 NewOpc = ARMISD::VLD2DUP_UPD;
15911 NumVecs = 2;
15912 break;
15913 case ARMISD::VLD3DUP:
15914 NewOpc = ARMISD::VLD3DUP_UPD;
15915 NumVecs = 3;
15916 break;
15917 case ARMISD::VLD4DUP:
15918 NewOpc = ARMISD::VLD4DUP_UPD;
15919 NumVecs = 4;
15920 break;
15921 case ISD::LOAD:
15922 NewOpc = ARMISD::VLD1_UPD;
15923 NumVecs = 1;
15924 isLaneOp = false;
15925 break;
15926 case ISD::STORE:
15927 NewOpc = ARMISD::VST1_UPD;
15928 NumVecs = 1;
15929 isLaneOp = false;
15930 isLoadOp = false;
15931 break;
15932 }
15933 }
15934
15935 // Find the size of memory referenced by the load/store.
15936 EVT VecTy;
15937 if (isLoadOp) {
15938 VecTy = N->getValueType(0);
15939 } else if (Target.isIntrinsic) {
15940 VecTy = N->getOperand(Target.AddrOpIdx + 1).getValueType();
15941 } else {
15942 assert(Target.isStore &&
15943 "Node has to be a load, a store, or an intrinsic!");
15944 VecTy = N->getOperand(1).getValueType();
15945 }
15946
15947 bool isVLDDUPOp =
15948 NewOpc == ARMISD::VLD1DUP_UPD || NewOpc == ARMISD::VLD2DUP_UPD ||
15949 NewOpc == ARMISD::VLD3DUP_UPD || NewOpc == ARMISD::VLD4DUP_UPD;
15950
15951 unsigned NumBytes = NumVecs * VecTy.getSizeInBits() / 8;
15952 if (isLaneOp || isVLDDUPOp)
15953 NumBytes /= VecTy.getVectorNumElements();
15954
15955 if (NumBytes >= 3 * 16 && User.ConstInc != NumBytes) {
15956 // VLD3/4 and VST3/4 for 128-bit vectors are implemented with two
15957 // separate instructions that make it harder to use a non-constant update.
15958 return false;
15959 }
15960
15961 if (SimpleConstIncOnly && User.ConstInc != NumBytes)
15962 return false;
15963
15964 if (!isValidBaseUpdate(N, User.N))
15965 return false;
15966
15967 // OK, we found an ADD we can fold into the base update.
15968 // Now, create a _UPD node, taking care of not breaking alignment.
15969
15970 EVT AlignedVecTy = VecTy;
15971 Align Alignment = MemN->getAlign();
15972
15973 // If this is a less-than-standard-aligned load/store, change the type to
15974 // match the standard alignment.
15975 // The alignment is overlooked when selecting _UPD variants; and it's
15976 // easier to introduce bitcasts here than fix that.
15977 // There are 3 ways to get to this base-update combine:
15978 // - intrinsics: they are assumed to be properly aligned (to the standard
15979 // alignment of the memory type), so we don't need to do anything.
15980 // - ARMISD::VLDx nodes: they are only generated from the aforementioned
15981 // intrinsics, so, likewise, there's nothing to do.
15982 // - generic load/store instructions: the alignment is specified as an
15983 // explicit operand, rather than implicitly as the standard alignment
15984 // of the memory type (like the intrinsics). We need to change the
15985 // memory type to match the explicit alignment. That way, we don't
15986 // generate non-standard-aligned ARMISD::VLDx nodes.
15987 if (isa<LSBaseSDNode>(N)) {
15988 if (Alignment.value() < VecTy.getScalarSizeInBits() / 8) {
15989 MVT EltTy = MVT::getIntegerVT(Alignment.value() * 8);
15990 assert(NumVecs == 1 && "Unexpected multi-element generic load/store.");
15991 assert(!isLaneOp && "Unexpected generic load/store lane.");
15992 unsigned NumElts = NumBytes / (EltTy.getSizeInBits() / 8);
15993 AlignedVecTy = MVT::getVectorVT(EltTy, NumElts);
15994 }
15995 // Don't set an explicit alignment on regular load/stores that we want
15996 // to transform to VLD/VST 1_UPD nodes.
15997 // This matches the behavior of regular load/stores, which only get an
15998 // explicit alignment if the MMO alignment is larger than the standard
15999 // alignment of the memory type.
16000 // Intrinsics, however, always get an explicit alignment, set to the
16001 // alignment of the MMO.
16002 Alignment = Align(1);
16003 }
16004
16005 // Create the new updating load/store node.
16006 // First, create an SDVTList for the new updating node's results.
16007 EVT Tys[6];
16008 unsigned NumResultVecs = (isLoadOp ? NumVecs : 0);
16009 unsigned n;
16010 for (n = 0; n < NumResultVecs; ++n)
16011 Tys[n] = AlignedVecTy;
16012 Tys[n++] = MVT::i32;
16013 Tys[n] = MVT::Other;
16014 SDVTList SDTys = DAG.getVTList(ArrayRef(Tys, NumResultVecs + 2));
16015
16016 // Then, gather the new node's operands.
16018 Ops.push_back(N->getOperand(0)); // incoming chain
16019 Ops.push_back(N->getOperand(Target.AddrOpIdx));
16020 Ops.push_back(User.Inc);
16021
16022 if (StoreSDNode *StN = dyn_cast<StoreSDNode>(N)) {
16023 // Try to match the intrinsic's signature
16024 Ops.push_back(StN->getValue());
16025 } else {
16026 // Loads (and of course intrinsics) match the intrinsics' signature,
16027 // so just add all but the alignment operand.
16028 unsigned LastOperand =
16029 hasAlignment ? N->getNumOperands() - 1 : N->getNumOperands();
16030 for (unsigned i = Target.AddrOpIdx + 1; i < LastOperand; ++i)
16031 Ops.push_back(N->getOperand(i));
16032 }
16033
16034 // For all node types, the alignment operand is always the last one.
16035 Ops.push_back(DAG.getConstant(Alignment.value(), dl, MVT::i32));
16036
16037 // If this is a non-standard-aligned STORE, the penultimate operand is the
16038 // stored value. Bitcast it to the aligned type.
16039 if (AlignedVecTy != VecTy && N->getOpcode() == ISD::STORE) {
16040 SDValue &StVal = Ops[Ops.size() - 2];
16041 StVal = DAG.getNode(ISD::BITCAST, dl, AlignedVecTy, StVal);
16042 }
16043
16044 EVT LoadVT = isLaneOp ? VecTy.getVectorElementType() : AlignedVecTy;
16045 SDValue UpdN = DAG.getMemIntrinsicNode(NewOpc, dl, SDTys, Ops, LoadVT,
16046 MemN->getMemOperand());
16047
16048 // Update the uses.
16049 SmallVector<SDValue, 5> NewResults;
16050 for (unsigned i = 0; i < NumResultVecs; ++i)
16051 NewResults.push_back(SDValue(UpdN.getNode(), i));
16052
16053 // If this is an non-standard-aligned LOAD, the first result is the loaded
16054 // value. Bitcast it to the expected result type.
16055 if (AlignedVecTy != VecTy && N->getOpcode() == ISD::LOAD) {
16056 SDValue &LdVal = NewResults[0];
16057 LdVal = DAG.getNode(ISD::BITCAST, dl, VecTy, LdVal);
16058 }
16059
16060 NewResults.push_back(SDValue(UpdN.getNode(), NumResultVecs + 1)); // chain
16061 DCI.CombineTo(N, NewResults);
16062 DCI.CombineTo(User.N, SDValue(UpdN.getNode(), NumResultVecs));
16063
16064 return true;
16065}
16066
16067// If (opcode ptr inc) is and ADD-like instruction, return the
16068// increment value. Otherwise return 0.
16069static unsigned getPointerConstIncrement(unsigned Opcode, SDValue Ptr,
16070 SDValue Inc, const SelectionDAG &DAG) {
16072 if (!CInc)
16073 return 0;
16074
16075 switch (Opcode) {
16076 case ARMISD::VLD1_UPD:
16077 case ISD::ADD:
16078 return CInc->getZExtValue();
16079 case ISD::OR: {
16080 if (DAG.haveNoCommonBitsSet(Ptr, Inc)) {
16081 // (OR ptr inc) is the same as (ADD ptr inc)
16082 return CInc->getZExtValue();
16083 }
16084 return 0;
16085 }
16086 default:
16087 return 0;
16088 }
16089}
16090
16092 switch (N->getOpcode()) {
16093 case ISD::ADD:
16094 case ISD::OR: {
16095 if (isa<ConstantSDNode>(N->getOperand(1))) {
16096 *Ptr = N->getOperand(0);
16097 *CInc = N->getOperand(1);
16098 return true;
16099 }
16100 return false;
16101 }
16102 case ARMISD::VLD1_UPD: {
16103 if (isa<ConstantSDNode>(N->getOperand(2))) {
16104 *Ptr = N->getOperand(1);
16105 *CInc = N->getOperand(2);
16106 return true;
16107 }
16108 return false;
16109 }
16110 default:
16111 return false;
16112 }
16113}
16114
16115/// CombineBaseUpdate - Target-specific DAG combine function for VLDDUP,
16116/// NEON load/store intrinsics, and generic vector load/stores, to merge
16117/// base address updates.
16118/// For generic load/stores, the memory type is assumed to be a vector.
16119/// The caller is assumed to have checked legality.
16122 const bool isIntrinsic = (N->getOpcode() == ISD::INTRINSIC_VOID ||
16123 N->getOpcode() == ISD::INTRINSIC_W_CHAIN);
16124 const bool isStore = N->getOpcode() == ISD::STORE;
16125 const unsigned AddrOpIdx = ((isIntrinsic || isStore) ? 2 : 1);
16126 BaseUpdateTarget Target = {N, isIntrinsic, isStore, AddrOpIdx};
16127
16128 // Limit the number of possible base-updates we look at to prevent degenerate
16129 // cases.
16130 unsigned MaxBaseUpdates = ArmMaxBaseUpdatesToCheck;
16131
16132 SDValue Addr = N->getOperand(AddrOpIdx);
16133
16135
16136 // Search for a use of the address operand that is an increment.
16137 for (SDUse &Use : Addr->uses()) {
16138 SDNode *User = Use.getUser();
16139 if (Use.getResNo() != Addr.getResNo() || User->getNumOperands() != 2)
16140 continue;
16141
16142 SDValue Inc = User->getOperand(Use.getOperandNo() == 1 ? 0 : 1);
16143 unsigned ConstInc =
16144 getPointerConstIncrement(User->getOpcode(), Addr, Inc, DCI.DAG);
16145
16146 if (ConstInc || User->getOpcode() == ISD::ADD) {
16147 BaseUpdates.push_back({User, Inc, ConstInc});
16148 if (BaseUpdates.size() >= MaxBaseUpdates)
16149 break;
16150 }
16151 }
16152
16153 // If the address is a constant pointer increment itself, find
16154 // another constant increment that has the same base operand
16155 SDValue Base;
16156 SDValue CInc;
16157 if (findPointerConstIncrement(Addr.getNode(), &Base, &CInc)) {
16158 unsigned Offset =
16159 getPointerConstIncrement(Addr->getOpcode(), Base, CInc, DCI.DAG);
16160 if (Offset) {
16161 for (SDUse &Use : Base->uses()) {
16162
16163 SDNode *User = Use.getUser();
16164 if (Use.getResNo() != Base.getResNo() || User == Addr.getNode() ||
16165 User->getNumOperands() != 2)
16166 continue;
16167
16168 SDValue UserInc = User->getOperand(Use.getOperandNo() == 0 ? 1 : 0);
16169 unsigned UserOffset =
16170 getPointerConstIncrement(User->getOpcode(), Base, UserInc, DCI.DAG);
16171
16172 if (!UserOffset || UserOffset <= Offset)
16173 continue;
16174
16175 unsigned NewConstInc = UserOffset - Offset;
16176 SDValue NewInc = DCI.DAG.getConstant(NewConstInc, SDLoc(N), MVT::i32);
16177 BaseUpdates.push_back({User, NewInc, NewConstInc});
16178 if (BaseUpdates.size() >= MaxBaseUpdates)
16179 break;
16180 }
16181 }
16182 }
16183
16184 // Try to fold the load/store with an update that matches memory
16185 // access size. This should work well for sequential loads.
16186 unsigned NumValidUpd = BaseUpdates.size();
16187 for (unsigned I = 0; I < NumValidUpd; I++) {
16188 BaseUpdateUser &User = BaseUpdates[I];
16189 if (TryCombineBaseUpdate(Target, User, /*SimpleConstIncOnly=*/true, DCI))
16190 return SDValue();
16191 }
16192
16193 // Try to fold with other users. Non-constant updates are considered
16194 // first, and constant updates are sorted to not break a sequence of
16195 // strided accesses (if there is any).
16196 llvm::stable_sort(BaseUpdates,
16197 [](const BaseUpdateUser &LHS, const BaseUpdateUser &RHS) {
16198 return LHS.ConstInc < RHS.ConstInc;
16199 });
16200 for (BaseUpdateUser &User : BaseUpdates) {
16201 if (TryCombineBaseUpdate(Target, User, /*SimpleConstIncOnly=*/false, DCI))
16202 return SDValue();
16203 }
16204 return SDValue();
16205}
16206
16209 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer())
16210 return SDValue();
16211
16212 return CombineBaseUpdate(N, DCI);
16213}
16214
16217 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer())
16218 return SDValue();
16219
16220 SelectionDAG &DAG = DCI.DAG;
16221 SDValue Addr = N->getOperand(2);
16222 MemSDNode *MemN = cast<MemSDNode>(N);
16223 SDLoc dl(N);
16224
16225 // For the stores, where there are multiple intrinsics we only actually want
16226 // to post-inc the last of the them.
16227 unsigned IntNo = N->getConstantOperandVal(1);
16228 if (IntNo == Intrinsic::arm_mve_vst2q && N->getConstantOperandVal(5) != 1)
16229 return SDValue();
16230 if (IntNo == Intrinsic::arm_mve_vst4q && N->getConstantOperandVal(7) != 3)
16231 return SDValue();
16232
16233 // Search for a use of the address operand that is an increment.
16234 for (SDUse &Use : Addr->uses()) {
16235 SDNode *User = Use.getUser();
16236 if (User->getOpcode() != ISD::ADD || Use.getResNo() != Addr.getResNo())
16237 continue;
16238
16239 // Check that the add is independent of the load/store. Otherwise, folding
16240 // it would create a cycle. We can avoid searching through Addr as it's a
16241 // predecessor to both.
16244 Visited.insert(Addr.getNode());
16245 Worklist.push_back(N);
16246 Worklist.push_back(User);
16247 const unsigned MaxSteps = 1024;
16248 if (SDNode::hasPredecessorHelper(N, Visited, Worklist, MaxSteps) ||
16249 SDNode::hasPredecessorHelper(User, Visited, Worklist, MaxSteps))
16250 continue;
16251
16252 // Find the new opcode for the updating load/store.
16253 bool isLoadOp = true;
16254 unsigned NewOpc = 0;
16255 unsigned NumVecs = 0;
16256 switch (IntNo) {
16257 default:
16258 llvm_unreachable("unexpected intrinsic for MVE VLDn combine");
16259 case Intrinsic::arm_mve_vld2q:
16260 NewOpc = ARMISD::VLD2_UPD;
16261 NumVecs = 2;
16262 break;
16263 case Intrinsic::arm_mve_vld4q:
16264 NewOpc = ARMISD::VLD4_UPD;
16265 NumVecs = 4;
16266 break;
16267 case Intrinsic::arm_mve_vst2q:
16268 NewOpc = ARMISD::VST2_UPD;
16269 NumVecs = 2;
16270 isLoadOp = false;
16271 break;
16272 case Intrinsic::arm_mve_vst4q:
16273 NewOpc = ARMISD::VST4_UPD;
16274 NumVecs = 4;
16275 isLoadOp = false;
16276 break;
16277 }
16278
16279 // Find the size of memory referenced by the load/store.
16280 EVT VecTy;
16281 if (isLoadOp) {
16282 VecTy = N->getValueType(0);
16283 } else {
16284 VecTy = N->getOperand(3).getValueType();
16285 }
16286
16287 unsigned NumBytes = NumVecs * VecTy.getSizeInBits() / 8;
16288
16289 // If the increment is a constant, it must match the memory ref size.
16290 SDValue Inc = User->getOperand(User->getOperand(0) == Addr ? 1 : 0);
16292 if (!CInc || CInc->getZExtValue() != NumBytes)
16293 continue;
16294
16295 // Create the new updating load/store node.
16296 // First, create an SDVTList for the new updating node's results.
16297 EVT Tys[6];
16298 unsigned NumResultVecs = (isLoadOp ? NumVecs : 0);
16299 unsigned n;
16300 for (n = 0; n < NumResultVecs; ++n)
16301 Tys[n] = VecTy;
16302 Tys[n++] = MVT::i32;
16303 Tys[n] = MVT::Other;
16304 SDVTList SDTys = DAG.getVTList(ArrayRef(Tys, NumResultVecs + 2));
16305
16306 // Then, gather the new node's operands.
16308 Ops.push_back(N->getOperand(0)); // incoming chain
16309 Ops.push_back(N->getOperand(2)); // ptr
16310 Ops.push_back(Inc);
16311
16312 for (unsigned i = 3; i < N->getNumOperands(); ++i)
16313 Ops.push_back(N->getOperand(i));
16314
16315 SDValue UpdN = DAG.getMemIntrinsicNode(NewOpc, dl, SDTys, Ops, VecTy,
16316 MemN->getMemOperand());
16317
16318 // Update the uses.
16319 SmallVector<SDValue, 5> NewResults;
16320 for (unsigned i = 0; i < NumResultVecs; ++i)
16321 NewResults.push_back(SDValue(UpdN.getNode(), i));
16322
16323 NewResults.push_back(SDValue(UpdN.getNode(), NumResultVecs + 1)); // chain
16324 DCI.CombineTo(N, NewResults);
16325 DCI.CombineTo(User, SDValue(UpdN.getNode(), NumResultVecs));
16326
16327 break;
16328 }
16329
16330 return SDValue();
16331}
16332
16333/// CombineVLDDUP - For a VDUPLANE node N, check if its source operand is a
16334/// vldN-lane (N > 1) intrinsic, and if all the other uses of that intrinsic
16335/// are also VDUPLANEs. If so, combine them to a vldN-dup operation and
16336/// return true.
16338 SelectionDAG &DAG = DCI.DAG;
16339 EVT VT = N->getValueType(0);
16340 // vldN-dup instructions only support 64-bit vectors for N > 1.
16341 if (!VT.is64BitVector())
16342 return false;
16343
16344 // Check if the VDUPLANE operand is a vldN-dup intrinsic.
16345 SDNode *VLD = N->getOperand(0).getNode();
16346 if (VLD->getOpcode() != ISD::INTRINSIC_W_CHAIN)
16347 return false;
16348 unsigned NumVecs = 0;
16349 unsigned NewOpc = 0;
16350 unsigned IntNo = VLD->getConstantOperandVal(1);
16351 if (IntNo == Intrinsic::arm_neon_vld2lane) {
16352 NumVecs = 2;
16353 NewOpc = ARMISD::VLD2DUP;
16354 } else if (IntNo == Intrinsic::arm_neon_vld3lane) {
16355 NumVecs = 3;
16356 NewOpc = ARMISD::VLD3DUP;
16357 } else if (IntNo == Intrinsic::arm_neon_vld4lane) {
16358 NumVecs = 4;
16359 NewOpc = ARMISD::VLD4DUP;
16360 } else {
16361 return false;
16362 }
16363
16364 // First check that all the vldN-lane uses are VDUPLANEs and that the lane
16365 // numbers match the load.
16366 unsigned VLDLaneNo = VLD->getConstantOperandVal(NumVecs + 3);
16367 for (SDUse &Use : VLD->uses()) {
16368 // Ignore uses of the chain result.
16369 if (Use.getResNo() == NumVecs)
16370 continue;
16371 SDNode *User = Use.getUser();
16372 if (User->getOpcode() != ARMISD::VDUPLANE ||
16373 VLDLaneNo != User->getConstantOperandVal(1))
16374 return false;
16375 }
16376
16377 // Create the vldN-dup node.
16378 EVT Tys[5];
16379 unsigned n;
16380 for (n = 0; n < NumVecs; ++n)
16381 Tys[n] = VT;
16382 Tys[n] = MVT::Other;
16383 SDVTList SDTys = DAG.getVTList(ArrayRef(Tys, NumVecs + 1));
16384 SDValue Ops[] = { VLD->getOperand(0), VLD->getOperand(2) };
16386 SDValue VLDDup = DAG.getMemIntrinsicNode(NewOpc, SDLoc(VLD), SDTys,
16387 Ops, VLDMemInt->getMemoryVT(),
16388 VLDMemInt->getMemOperand());
16389
16390 // Update the uses.
16391 for (SDUse &Use : VLD->uses()) {
16392 unsigned ResNo = Use.getResNo();
16393 // Ignore uses of the chain result.
16394 if (ResNo == NumVecs)
16395 continue;
16396 DCI.CombineTo(Use.getUser(), SDValue(VLDDup.getNode(), ResNo));
16397 }
16398
16399 // Now the vldN-lane intrinsic is dead except for its chain result.
16400 // Update uses of the chain.
16401 std::vector<SDValue> VLDDupResults;
16402 for (unsigned n = 0; n < NumVecs; ++n)
16403 VLDDupResults.push_back(SDValue(VLDDup.getNode(), n));
16404 VLDDupResults.push_back(SDValue(VLDDup.getNode(), NumVecs));
16405 DCI.CombineTo(VLD, VLDDupResults);
16406
16407 return true;
16408}
16409
16410/// PerformVDUPLANECombine - Target-specific dag combine xforms for
16411/// ARMISD::VDUPLANE.
16414 const ARMSubtarget *Subtarget) {
16415 SDValue Op = N->getOperand(0);
16416 EVT VT = N->getValueType(0);
16417
16418 // On MVE, we just convert the VDUPLANE to a VDUP with an extract.
16419 if (Subtarget->hasMVEIntegerOps()) {
16420 EVT ExtractVT = VT.getVectorElementType();
16421 // We need to ensure we are creating a legal type.
16422 if (!DCI.DAG.getTargetLoweringInfo().isTypeLegal(ExtractVT))
16423 ExtractVT = MVT::i32;
16424 SDValue Extract = DCI.DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), ExtractVT,
16425 N->getOperand(0), N->getOperand(1));
16426 return DCI.DAG.getNode(ARMISD::VDUP, SDLoc(N), VT, Extract);
16427 }
16428
16429 // If the source is a vldN-lane (N > 1) intrinsic, and all the other uses
16430 // of that intrinsic are also VDUPLANEs, combine them to a vldN-dup operation.
16431 if (CombineVLDDUP(N, DCI))
16432 return SDValue(N, 0);
16433
16434 // If the source is already a VMOVIMM or VMVNIMM splat, the VDUPLANE is
16435 // redundant. Ignore bit_converts for now; element sizes are checked below.
16436 while (Op.getOpcode() == ISD::BITCAST)
16437 Op = Op.getOperand(0);
16438 if (Op.getOpcode() != ARMISD::VMOVIMM && Op.getOpcode() != ARMISD::VMVNIMM)
16439 return SDValue();
16440
16441 // Make sure the VMOV element size is not bigger than the VDUPLANE elements.
16442 unsigned EltSize = Op.getScalarValueSizeInBits();
16443 // The canonical VMOV for a zero vector uses a 32-bit element size.
16444 unsigned Imm = Op.getConstantOperandVal(0);
16445 unsigned EltBits;
16446 if (ARM_AM::decodeVMOVModImm(Imm, EltBits) == 0)
16447 EltSize = 8;
16448 if (EltSize > VT.getScalarSizeInBits())
16449 return SDValue();
16450
16451 return DCI.DAG.getNode(ISD::BITCAST, SDLoc(N), VT, Op);
16452}
16453
16454/// PerformVDUPCombine - Target-specific dag combine xforms for ARMISD::VDUP.
16456 const ARMSubtarget *Subtarget) {
16457 SDValue Op = N->getOperand(0);
16458 SDLoc dl(N);
16459
16460 if (Subtarget->hasMVEIntegerOps()) {
16461 // Convert VDUP f32 -> VDUP BITCAST i32 under MVE, as we know the value will
16462 // need to come from a GPR.
16463 if (Op.getValueType() == MVT::f32)
16464 return DAG.getNode(ARMISD::VDUP, dl, N->getValueType(0),
16465 DAG.getNode(ISD::BITCAST, dl, MVT::i32, Op));
16466 else if (Op.getValueType() == MVT::f16)
16467 return DAG.getNode(ARMISD::VDUP, dl, N->getValueType(0),
16468 DAG.getNode(ARMISD::VMOVrh, dl, MVT::i32, Op));
16469 }
16470
16471 if (!Subtarget->hasNEON())
16472 return SDValue();
16473
16474 // Match VDUP(LOAD) -> VLD1DUP.
16475 // We match this pattern here rather than waiting for isel because the
16476 // transform is only legal for unindexed loads.
16477 LoadSDNode *LD = dyn_cast<LoadSDNode>(Op.getNode());
16478 if (LD && Op.hasOneUse() && LD->isUnindexed() &&
16479 LD->getMemoryVT() == N->getValueType(0).getVectorElementType()) {
16480 SDValue Ops[] = {LD->getOperand(0), LD->getOperand(1),
16481 DAG.getConstant(LD->getAlign().value(), SDLoc(N), MVT::i32)};
16482 SDVTList SDTys = DAG.getVTList(N->getValueType(0), MVT::Other);
16483 SDValue VLDDup =
16485 LD->getMemoryVT(), LD->getMemOperand());
16486 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), VLDDup.getValue(1));
16487 return VLDDup;
16488 }
16489
16490 return SDValue();
16491}
16492
16495 const ARMSubtarget *Subtarget) {
16496 EVT VT = N->getValueType(0);
16497
16498 // If this is a legal vector load, try to combine it into a VLD1_UPD.
16499 if (Subtarget->hasNEON() && ISD::isNormalLoad(N) && VT.isVector() &&
16501 return CombineBaseUpdate(N, DCI);
16502
16503 return SDValue();
16504}
16505
16506// Optimize trunc store (of multiple scalars) to shuffle and store. First,
16507// pack all of the elements in one place. Next, store to memory in fewer
16508// chunks.
16510 SelectionDAG &DAG) {
16511 SDValue StVal = St->getValue();
16512 EVT VT = StVal.getValueType();
16513 if (!St->isTruncatingStore() || !VT.isVector())
16514 return SDValue();
16515 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
16516 EVT StVT = St->getMemoryVT();
16517 unsigned NumElems = VT.getVectorNumElements();
16518 assert(StVT != VT && "Cannot truncate to the same type");
16519 unsigned FromEltSz = VT.getScalarSizeInBits();
16520 unsigned ToEltSz = StVT.getScalarSizeInBits();
16521
16522 // From, To sizes and ElemCount must be pow of two
16523 if (!isPowerOf2_32(NumElems * FromEltSz * ToEltSz))
16524 return SDValue();
16525
16526 // We are going to use the original vector elt for storing.
16527 // Accumulated smaller vector elements must be a multiple of the store size.
16528 if (0 != (NumElems * FromEltSz) % ToEltSz)
16529 return SDValue();
16530
16531 unsigned SizeRatio = FromEltSz / ToEltSz;
16532 assert(SizeRatio * NumElems * ToEltSz == VT.getSizeInBits());
16533
16534 // Create a type on which we perform the shuffle.
16535 EVT WideVecVT = EVT::getVectorVT(*DAG.getContext(), StVT.getScalarType(),
16536 NumElems * SizeRatio);
16537 assert(WideVecVT.getSizeInBits() == VT.getSizeInBits());
16538
16539 SDLoc DL(St);
16540 SDValue WideVec = DAG.getNode(ISD::BITCAST, DL, WideVecVT, StVal);
16541 SmallVector<int, 8> ShuffleVec(NumElems * SizeRatio, -1);
16542 for (unsigned i = 0; i < NumElems; ++i)
16543 ShuffleVec[i] = DAG.getDataLayout().isBigEndian() ? (i + 1) * SizeRatio - 1
16544 : i * SizeRatio;
16545
16546 // Can't shuffle using an illegal type.
16547 if (!TLI.isTypeLegal(WideVecVT))
16548 return SDValue();
16549
16550 SDValue Shuff = DAG.getVectorShuffle(
16551 WideVecVT, DL, WideVec, DAG.getUNDEF(WideVec.getValueType()), ShuffleVec);
16552 // At this point all of the data is stored at the bottom of the
16553 // register. We now need to save it to mem.
16554
16555 // Find the largest store unit
16556 MVT StoreType = MVT::i8;
16557 for (MVT Tp : MVT::integer_valuetypes()) {
16558 if (TLI.isTypeLegal(Tp) && Tp.getSizeInBits() <= NumElems * ToEltSz)
16559 StoreType = Tp;
16560 }
16561 // Didn't find a legal store type.
16562 if (!TLI.isTypeLegal(StoreType))
16563 return SDValue();
16564
16565 // Bitcast the original vector into a vector of store-size units
16566 EVT StoreVecVT =
16567 EVT::getVectorVT(*DAG.getContext(), StoreType,
16568 VT.getSizeInBits() / EVT(StoreType).getSizeInBits());
16569 assert(StoreVecVT.getSizeInBits() == VT.getSizeInBits());
16570 SDValue ShuffWide = DAG.getNode(ISD::BITCAST, DL, StoreVecVT, Shuff);
16572 SDValue Increment = DAG.getConstant(StoreType.getSizeInBits() / 8, DL,
16573 TLI.getPointerTy(DAG.getDataLayout()));
16574 SDValue BasePtr = St->getBasePtr();
16575
16576 // Perform one or more big stores into memory.
16577 unsigned E = (ToEltSz * NumElems) / StoreType.getSizeInBits();
16578 for (unsigned I = 0; I < E; I++) {
16579 SDValue SubVec = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, StoreType,
16580 ShuffWide, DAG.getIntPtrConstant(I, DL));
16581 SDValue Ch =
16582 DAG.getStore(St->getChain(), DL, SubVec, BasePtr, St->getPointerInfo(),
16583 St->getAlign(), St->getMemOperand()->getFlags());
16584 BasePtr =
16585 DAG.getNode(ISD::ADD, DL, BasePtr.getValueType(), BasePtr, Increment);
16586 Chains.push_back(Ch);
16587 }
16588 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
16589}
16590
16591// Try taking a single vector store from an fpround (which would otherwise turn
16592// into an expensive buildvector) and splitting it into a series of narrowing
16593// stores.
16595 SelectionDAG &DAG) {
16596 if (!St->isSimple() || St->isTruncatingStore() || !St->isUnindexed())
16597 return SDValue();
16598 SDValue Trunc = St->getValue();
16599 if (Trunc->getOpcode() != ISD::FP_ROUND)
16600 return SDValue();
16601 EVT FromVT = Trunc->getOperand(0).getValueType();
16602 EVT ToVT = Trunc.getValueType();
16603 if (!ToVT.isVector())
16604 return SDValue();
16606 EVT ToEltVT = ToVT.getVectorElementType();
16607 EVT FromEltVT = FromVT.getVectorElementType();
16608
16609 if (FromEltVT != MVT::f32 || ToEltVT != MVT::f16)
16610 return SDValue();
16611
16612 unsigned NumElements = 4;
16613 if (FromVT.getVectorNumElements() % NumElements != 0)
16614 return SDValue();
16615
16616 // Test if the Trunc will be convertible to a VMOVN with a shuffle, and if so
16617 // use the VMOVN over splitting the store. We are looking for patterns of:
16618 // !rev: 0 N 1 N+1 2 N+2 ...
16619 // rev: N 0 N+1 1 N+2 2 ...
16620 // The shuffle may either be a single source (in which case N = NumElts/2) or
16621 // two inputs extended with concat to the same size (in which case N =
16622 // NumElts).
16623 auto isVMOVNShuffle = [&](ShuffleVectorSDNode *SVN, bool Rev) {
16624 ArrayRef<int> M = SVN->getMask();
16625 unsigned NumElts = ToVT.getVectorNumElements();
16626 if (SVN->getOperand(1).isUndef())
16627 NumElts /= 2;
16628
16629 unsigned Off0 = Rev ? NumElts : 0;
16630 unsigned Off1 = Rev ? 0 : NumElts;
16631
16632 for (unsigned I = 0; I < NumElts; I += 2) {
16633 if (M[I] >= 0 && M[I] != (int)(Off0 + I / 2))
16634 return false;
16635 if (M[I + 1] >= 0 && M[I + 1] != (int)(Off1 + I / 2))
16636 return false;
16637 }
16638
16639 return true;
16640 };
16641
16642 if (auto *Shuffle = dyn_cast<ShuffleVectorSDNode>(Trunc.getOperand(0)))
16643 if (isVMOVNShuffle(Shuffle, false) || isVMOVNShuffle(Shuffle, true))
16644 return SDValue();
16645
16646 LLVMContext &C = *DAG.getContext();
16647 SDLoc DL(St);
16648 // Details about the old store
16649 SDValue Ch = St->getChain();
16650 SDValue BasePtr = St->getBasePtr();
16651 Align Alignment = St->getBaseAlign();
16652 MachineMemOperand::Flags MMOFlags = St->getMemOperand()->getFlags();
16653 AAMDNodes AAInfo = St->getAAInfo();
16654
16655 // We split the store into slices of NumElements. fp16 trunc stores are vcvt
16656 // and then stored as truncating integer stores.
16657 EVT NewFromVT = EVT::getVectorVT(C, FromEltVT, NumElements);
16658 EVT NewToVT = EVT::getVectorVT(
16659 C, EVT::getIntegerVT(C, ToEltVT.getSizeInBits()), NumElements);
16660
16662 for (unsigned i = 0; i < FromVT.getVectorNumElements() / NumElements; i++) {
16663 unsigned NewOffset = i * NumElements * ToEltVT.getSizeInBits() / 8;
16664 SDValue NewPtr =
16665 DAG.getObjectPtrOffset(DL, BasePtr, TypeSize::getFixed(NewOffset));
16666
16667 SDValue Extract =
16668 DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, NewFromVT, Trunc.getOperand(0),
16669 DAG.getConstant(i * NumElements, DL, MVT::i32));
16670
16671 SDValue FPTrunc =
16672 DAG.getNode(ARMISD::VCVTN, DL, MVT::v8f16, DAG.getUNDEF(MVT::v8f16),
16673 Extract, DAG.getConstant(0, DL, MVT::i32));
16674 Extract = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, MVT::v4i32, FPTrunc);
16675
16677 Ch, DL, Extract, NewPtr, St->getPointerInfo().getWithOffset(NewOffset),
16678 NewToVT, Alignment, MMOFlags, AAInfo);
16679 Stores.push_back(Store);
16680 }
16681 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Stores);
16682}
16683
16684// Try taking a single vector store from an MVETRUNC (which would otherwise turn
16685// into an expensive buildvector) and splitting it into a series of narrowing
16686// stores.
16688 SelectionDAG &DAG) {
16689 if (!St->isSimple() || St->isTruncatingStore() || !St->isUnindexed())
16690 return SDValue();
16691 SDValue Trunc = St->getValue();
16692 if (Trunc->getOpcode() != ARMISD::MVETRUNC)
16693 return SDValue();
16694 EVT FromVT = Trunc->getOperand(0).getValueType();
16695 EVT ToVT = Trunc.getValueType();
16696
16697 LLVMContext &C = *DAG.getContext();
16698 SDLoc DL(St);
16699 // Details about the old store
16700 SDValue Ch = St->getChain();
16701 SDValue BasePtr = St->getBasePtr();
16702 Align Alignment = St->getBaseAlign();
16703 MachineMemOperand::Flags MMOFlags = St->getMemOperand()->getFlags();
16704 AAMDNodes AAInfo = St->getAAInfo();
16705
16706 EVT NewToVT = EVT::getVectorVT(C, ToVT.getVectorElementType(),
16707 FromVT.getVectorNumElements());
16708
16710 for (unsigned i = 0; i < Trunc.getNumOperands(); i++) {
16711 unsigned NewOffset =
16712 i * FromVT.getVectorNumElements() * ToVT.getScalarSizeInBits() / 8;
16713 SDValue NewPtr =
16714 DAG.getObjectPtrOffset(DL, BasePtr, TypeSize::getFixed(NewOffset));
16715
16716 SDValue Extract = Trunc.getOperand(i);
16718 Ch, DL, Extract, NewPtr, St->getPointerInfo().getWithOffset(NewOffset),
16719 NewToVT, Alignment, MMOFlags, AAInfo);
16720 Stores.push_back(Store);
16721 }
16722 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Stores);
16723}
16724
16725// Given a floating point store from an extracted vector, with an integer
16726// VGETLANE that already exists, store the existing VGETLANEu directly. This can
16727// help reduce fp register pressure, doesn't require the fp extract and allows
16728// use of more integer post-inc stores not available with vstr.
16730 if (!St->isSimple() || St->isTruncatingStore() || !St->isUnindexed())
16731 return SDValue();
16732 SDValue Extract = St->getValue();
16733 EVT VT = Extract.getValueType();
16734 // For now only uses f16. This may be useful for f32 too, but that will
16735 // be bitcast(extract), not the VGETLANEu we currently check here.
16736 if (VT != MVT::f16 || Extract->getOpcode() != ISD::EXTRACT_VECTOR_ELT)
16737 return SDValue();
16738
16739 SDNode *GetLane =
16740 DAG.getNodeIfExists(ARMISD::VGETLANEu, DAG.getVTList(MVT::i32),
16741 {Extract.getOperand(0), Extract.getOperand(1)});
16742 if (!GetLane)
16743 return SDValue();
16744
16745 LLVMContext &C = *DAG.getContext();
16746 SDLoc DL(St);
16747 // Create a new integer store to replace the existing floating point version.
16748 SDValue Ch = St->getChain();
16749 SDValue BasePtr = St->getBasePtr();
16750 Align Alignment = St->getBaseAlign();
16751 MachineMemOperand::Flags MMOFlags = St->getMemOperand()->getFlags();
16752 AAMDNodes AAInfo = St->getAAInfo();
16753 EVT NewToVT = EVT::getIntegerVT(C, VT.getSizeInBits());
16754 SDValue Store = DAG.getTruncStore(Ch, DL, SDValue(GetLane, 0), BasePtr,
16755 St->getPointerInfo(), NewToVT, Alignment,
16756 MMOFlags, AAInfo);
16757
16758 return Store;
16759}
16760
16761/// PerformSTORECombine - Target-specific dag combine xforms for
16762/// ISD::STORE.
16765 const ARMSubtarget *Subtarget) {
16767 if (St->isVolatile())
16768 return SDValue();
16769 SDValue StVal = St->getValue();
16770 EVT VT = StVal.getValueType();
16771
16772 if (Subtarget->hasNEON())
16774 return Store;
16775
16776 if (Subtarget->hasMVEFloatOps())
16777 if (SDValue NewToken = PerformSplittingToNarrowingStores(St, DCI.DAG))
16778 return NewToken;
16779
16780 if (Subtarget->hasMVEIntegerOps()) {
16781 if (SDValue NewChain = PerformExtractFpToIntStores(St, DCI.DAG))
16782 return NewChain;
16783 if (SDValue NewToken =
16785 return NewToken;
16786 }
16787
16788 if (!ISD::isNormalStore(St))
16789 return SDValue();
16790
16791 // Split a store of a VMOVDRR into two integer stores to avoid mixing NEON and
16792 // ARM stores of arguments in the same cache line.
16793 if (StVal.getOpcode() == ARMISD::VMOVDRR && StVal->hasOneUse()) {
16794 SelectionDAG &DAG = DCI.DAG;
16795 bool isBigEndian = DAG.getDataLayout().isBigEndian();
16796 SDLoc DL(St);
16797 SDValue BasePtr = St->getBasePtr();
16798 SDValue NewST1 =
16799 DAG.getStore(St->getChain(), DL, StVal.getOperand(isBigEndian ? 1 : 0),
16800 BasePtr, St->getPointerInfo(), St->getBaseAlign(),
16801 St->getMemOperand()->getFlags());
16802
16803 SDValue OffsetPtr = DAG.getNode(ISD::ADD, DL, MVT::i32, BasePtr,
16804 DAG.getConstant(4, DL, MVT::i32));
16805 return DAG.getStore(NewST1.getValue(0), DL,
16806 StVal.getOperand(isBigEndian ? 0 : 1), OffsetPtr,
16807 St->getPointerInfo().getWithOffset(4),
16808 St->getBaseAlign(), St->getMemOperand()->getFlags());
16809 }
16810
16811 if (StVal.getValueType() == MVT::i64 &&
16813 // Bitcast an i64 store extracted from a vector to f64.
16814 // Otherwise, the i64 value will be legalized to a pair of i32 values.
16815 SelectionDAG &DAG = DCI.DAG;
16816 SDLoc dl(StVal);
16817 SDValue IntVec = StVal.getOperand(0);
16818 EVT FloatVT =
16819 EVT::getVectorVT(*DAG.getContext(), MVT::f64,
16821 SDValue Vec = DAG.getNode(ISD::BITCAST, dl, FloatVT, IntVec);
16822 SDValue ExtElt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, dl, MVT::f64, Vec,
16823 StVal.getOperand(1));
16824 dl = SDLoc(N);
16825 SDValue V = DAG.getNode(ISD::BITCAST, dl, MVT::i64, ExtElt);
16826 // Make the DAGCombiner fold the bitcasts.
16827 DCI.AddToWorklist(Vec.getNode());
16828 DCI.AddToWorklist(ExtElt.getNode());
16829 DCI.AddToWorklist(V.getNode());
16830 return DAG.getStore(St->getChain(), dl, V, St->getBasePtr(),
16831 St->getPointerInfo(), St->getAlign(),
16832 St->getMemOperand()->getFlags(), St->getAAInfo());
16833 }
16834
16835 // If this is a legal vector store, try to combine it into a VST1_UPD.
16836 if (Subtarget->hasNEON() && ISD::isNormalStore(N) && VT.isVector() &&
16838 return CombineBaseUpdate(N, DCI);
16839
16840 return SDValue();
16841}
16842
16843/// PerformVCVTCombine - VCVT (floating-point to fixed-point, Advanced SIMD)
16844/// can replace combinations of VMUL and VCVT (floating-point to integer)
16845/// when the VMUL has a constant operand that is a power of 2.
16846///
16847/// Example (assume d17 = <float 8.000000e+00, float 8.000000e+00>):
16848/// vmul.f32 d16, d17, d16
16849/// vcvt.s32.f32 d16, d16
16850/// becomes:
16851/// vcvt.s32.f32 d16, d16, #3
16853 const ARMSubtarget *Subtarget) {
16854 if (!Subtarget->hasNEON())
16855 return SDValue();
16856
16857 SDValue Op = N->getOperand(0);
16858 if (!Op.getValueType().isVector() || !Op.getValueType().isSimple() ||
16859 Op.getOpcode() != ISD::FMUL)
16860 return SDValue();
16861
16862 SDValue ConstVec = Op->getOperand(1);
16863 if (!isa<BuildVectorSDNode>(ConstVec))
16864 return SDValue();
16865
16866 MVT FloatTy = Op.getSimpleValueType().getVectorElementType();
16867 uint32_t FloatBits = FloatTy.getSizeInBits();
16868 MVT IntTy = N->getSimpleValueType(0).getVectorElementType();
16869 uint32_t IntBits = IntTy.getSizeInBits();
16870 unsigned NumLanes = Op.getValueType().getVectorNumElements();
16871 if (FloatBits != 32 || IntBits > 32 || (NumLanes != 4 && NumLanes != 2)) {
16872 // These instructions only exist converting from f32 to i32. We can handle
16873 // smaller integers by generating an extra truncate, but larger ones would
16874 // be lossy. We also can't handle anything other than 2 or 4 lanes, since
16875 // these instructions only support v2i32/v4i32 types.
16876 return SDValue();
16877 }
16878
16879 BitVector UndefElements;
16881 int32_t C = BV->getConstantFPSplatPow2ToLog2Int(&UndefElements, 33);
16882 if (C == -1 || C == 0 || C > 32)
16883 return SDValue();
16884
16885 SDLoc dl(N);
16886 bool isSigned = N->getOpcode() == ISD::FP_TO_SINT;
16887 unsigned IntrinsicOpcode = isSigned ? Intrinsic::arm_neon_vcvtfp2fxs :
16888 Intrinsic::arm_neon_vcvtfp2fxu;
16889 SDValue FixConv = DAG.getNode(
16890 ISD::INTRINSIC_WO_CHAIN, dl, NumLanes == 2 ? MVT::v2i32 : MVT::v4i32,
16891 DAG.getConstant(IntrinsicOpcode, dl, MVT::i32), Op->getOperand(0),
16892 DAG.getConstant(C, dl, MVT::i32));
16893
16894 if (IntBits < FloatBits)
16895 FixConv = DAG.getNode(ISD::TRUNCATE, dl, N->getValueType(0), FixConv);
16896
16897 return FixConv;
16898}
16899
16901 const ARMSubtarget *Subtarget) {
16902 if (!Subtarget->hasMVEFloatOps())
16903 return SDValue();
16904
16905 // Turn (fadd x, (vselect c, y, -0.0)) into (vselect c, (fadd x, y), x)
16906 // The second form can be more easily turned into a predicated vadd, and
16907 // possibly combined into a fma to become a predicated vfma.
16908 SDValue Op0 = N->getOperand(0);
16909 SDValue Op1 = N->getOperand(1);
16910 EVT VT = N->getValueType(0);
16911 SDLoc DL(N);
16912
16913 // The identity element for a fadd is -0.0 or +0.0 when the nsz flag is set,
16914 // which these VMOV's represent.
16915 auto isIdentitySplat = [&](SDValue Op, bool NSZ) {
16916 if (Op.getOpcode() != ISD::BITCAST ||
16917 Op.getOperand(0).getOpcode() != ARMISD::VMOVIMM)
16918 return false;
16919 uint64_t ImmVal = Op.getOperand(0).getConstantOperandVal(0);
16920 if (VT == MVT::v4f32 && (ImmVal == 1664 || (ImmVal == 0 && NSZ)))
16921 return true;
16922 if (VT == MVT::v8f16 && (ImmVal == 2688 || (ImmVal == 0 && NSZ)))
16923 return true;
16924 return false;
16925 };
16926
16927 if (Op0.getOpcode() == ISD::VSELECT && Op1.getOpcode() != ISD::VSELECT)
16928 std::swap(Op0, Op1);
16929
16930 if (Op1.getOpcode() != ISD::VSELECT)
16931 return SDValue();
16932
16933 SDNodeFlags FaddFlags = N->getFlags();
16934 bool NSZ = FaddFlags.hasNoSignedZeros();
16935 if (!isIdentitySplat(Op1.getOperand(2), NSZ))
16936 return SDValue();
16937
16938 SDValue FAdd =
16939 DAG.getNode(ISD::FADD, DL, VT, Op0, Op1.getOperand(1), FaddFlags);
16940 return DAG.getNode(ISD::VSELECT, DL, VT, Op1.getOperand(0), FAdd, Op0, FaddFlags);
16941}
16942
16944 SDValue LHS = N->getOperand(0);
16945 SDValue RHS = N->getOperand(1);
16946 EVT VT = N->getValueType(0);
16947 SDLoc DL(N);
16948
16949 if (!N->getFlags().hasAllowReassociation())
16950 return SDValue();
16951
16952 // Combine fadd(a, vcmla(b, c, d)) -> vcmla(fadd(a, b), b, c)
16953 auto ReassocComplex = [&](SDValue A, SDValue B) {
16954 if (A.getOpcode() != ISD::INTRINSIC_WO_CHAIN)
16955 return SDValue();
16956 unsigned Opc = A.getConstantOperandVal(0);
16957 if (Opc != Intrinsic::arm_mve_vcmlaq)
16958 return SDValue();
16959 SDValue VCMLA = DAG.getNode(
16960 ISD::INTRINSIC_WO_CHAIN, DL, VT, A.getOperand(0), A.getOperand(1),
16961 DAG.getNode(ISD::FADD, DL, VT, A.getOperand(2), B, N->getFlags()),
16962 A.getOperand(3), A.getOperand(4));
16963 VCMLA->setFlags(A->getFlags());
16964 return VCMLA;
16965 };
16966 if (SDValue R = ReassocComplex(LHS, RHS))
16967 return R;
16968 if (SDValue R = ReassocComplex(RHS, LHS))
16969 return R;
16970
16971 return SDValue();
16972}
16973
16975 const ARMSubtarget *Subtarget) {
16976 if (SDValue S = PerformFAddVSelectCombine(N, DAG, Subtarget))
16977 return S;
16978 if (SDValue S = PerformFADDVCMLACombine(N, DAG))
16979 return S;
16980 return SDValue();
16981}
16982
16983/// PerformVMulVCTPCombine - VCVT (fixed-point to floating-point, Advanced SIMD)
16984/// can replace combinations of VCVT (integer to floating-point) and VMUL
16985/// when the VMUL has a constant operand that is a power of 2.
16986///
16987/// Example (assume d17 = <float 0.125, float 0.125>):
16988/// vcvt.f32.s32 d16, d16
16989/// vmul.f32 d16, d16, d17
16990/// becomes:
16991/// vcvt.f32.s32 d16, d16, #3
16993 const ARMSubtarget *Subtarget) {
16994 if (!Subtarget->hasNEON())
16995 return SDValue();
16996
16997 SDValue Op = N->getOperand(0);
16998 unsigned OpOpcode = Op.getNode()->getOpcode();
16999 if (!N->getValueType(0).isVector() || !N->getValueType(0).isSimple() ||
17000 (OpOpcode != ISD::SINT_TO_FP && OpOpcode != ISD::UINT_TO_FP))
17001 return SDValue();
17002
17003 SDValue ConstVec = N->getOperand(1);
17004 if (!isa<BuildVectorSDNode>(ConstVec))
17005 return SDValue();
17006
17007 MVT FloatTy = N->getSimpleValueType(0).getVectorElementType();
17008 uint32_t FloatBits = FloatTy.getSizeInBits();
17009 MVT IntTy = Op.getOperand(0).getSimpleValueType().getVectorElementType();
17010 uint32_t IntBits = IntTy.getSizeInBits();
17011 unsigned NumLanes = Op.getValueType().getVectorNumElements();
17012 if (FloatBits != 32 || IntBits > 32 || (NumLanes != 4 && NumLanes != 2)) {
17013 // These instructions only exist converting from i32 to f32. We can handle
17014 // smaller integers by generating an extra extend, but larger ones would
17015 // be lossy. We also can't handle anything other than 2 or 4 lanes, since
17016 // these instructions only support v2i32/v4i32 types.
17017 return SDValue();
17018 }
17019
17020 ConstantFPSDNode *CN = isConstOrConstSplatFP(ConstVec, true);
17021 APFloat Recip(0.0f);
17022 if (!CN || !CN->getValueAPF().getExactInverse(&Recip))
17023 return SDValue();
17024
17025 bool IsExact;
17026 APSInt IntVal(33);
17027 if (Recip.convertToInteger(IntVal, APFloat::rmTowardZero, &IsExact) !=
17028 APFloat::opOK ||
17029 !IsExact)
17030 return SDValue();
17031
17032 int32_t C = IntVal.exactLogBase2();
17033 if (C == -1 || C == 0 || C > 32)
17034 return SDValue();
17035
17036 SDLoc DL(N);
17037 bool isSigned = OpOpcode == ISD::SINT_TO_FP;
17038 SDValue ConvInput = Op.getOperand(0);
17039 if (IntBits < FloatBits)
17041 NumLanes == 2 ? MVT::v2i32 : MVT::v4i32, ConvInput);
17042
17043 unsigned IntrinsicOpcode = isSigned ? Intrinsic::arm_neon_vcvtfxs2fp
17044 : Intrinsic::arm_neon_vcvtfxu2fp;
17045 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, Op.getValueType(),
17046 DAG.getConstant(IntrinsicOpcode, DL, MVT::i32), ConvInput,
17047 DAG.getConstant(C, DL, MVT::i32));
17048}
17049
17051 const ARMSubtarget *ST) {
17052 if (!ST->hasMVEIntegerOps())
17053 return SDValue();
17054
17055 assert(N->getOpcode() == ISD::VECREDUCE_ADD);
17056 EVT ResVT = N->getValueType(0);
17057 SDValue N0 = N->getOperand(0);
17058 SDLoc dl(N);
17059
17060 // Try to turn vecreduce_add(add(x, y)) into vecreduce(x) + vecreduce(y)
17061 if (ResVT == MVT::i32 && N0.getOpcode() == ISD::ADD &&
17062 (N0.getValueType() == MVT::v4i32 || N0.getValueType() == MVT::v8i16 ||
17063 N0.getValueType() == MVT::v16i8)) {
17064 SDValue Red0 = DAG.getNode(ISD::VECREDUCE_ADD, dl, ResVT, N0.getOperand(0));
17065 SDValue Red1 = DAG.getNode(ISD::VECREDUCE_ADD, dl, ResVT, N0.getOperand(1));
17066 return DAG.getNode(ISD::ADD, dl, ResVT, Red0, Red1);
17067 }
17068
17069 // We are looking for something that will have illegal types if left alone,
17070 // but that we can convert to a single instruction under MVE. For example
17071 // vecreduce_add(sext(A, v8i32)) => VADDV.s16 A
17072 // or
17073 // vecreduce_add(mul(zext(A, v16i32), zext(B, v16i32))) => VMLADAV.u8 A, B
17074
17075 // The legal cases are:
17076 // VADDV u/s 8/16/32
17077 // VMLAV u/s 8/16/32
17078 // VADDLV u/s 32
17079 // VMLALV u/s 16/32
17080
17081 // If the input vector is smaller than legal (v4i8/v4i16 for example) we can
17082 // extend it and use v4i32 instead.
17083 auto ExtTypeMatches = [](SDValue A, ArrayRef<MVT> ExtTypes) {
17084 EVT AVT = A.getValueType();
17085 return any_of(ExtTypes, [&](MVT Ty) {
17086 return AVT.getVectorNumElements() == Ty.getVectorNumElements() &&
17087 AVT.bitsLE(Ty);
17088 });
17089 };
17090 auto ExtendIfNeeded = [&](SDValue A, unsigned ExtendCode) {
17091 EVT AVT = A.getValueType();
17092 if (!AVT.is128BitVector())
17093 A = DAG.getNode(
17094 ExtendCode, dl,
17096 *DAG.getContext(),
17098 A);
17099 return A;
17100 };
17101 auto IsVADDV = [&](MVT RetTy, unsigned ExtendCode, ArrayRef<MVT> ExtTypes) {
17102 if (ResVT != RetTy || N0->getOpcode() != ExtendCode)
17103 return SDValue();
17104 SDValue A = N0->getOperand(0);
17105 if (ExtTypeMatches(A, ExtTypes))
17106 return ExtendIfNeeded(A, ExtendCode);
17107 return SDValue();
17108 };
17109 auto IsPredVADDV = [&](MVT RetTy, unsigned ExtendCode,
17110 ArrayRef<MVT> ExtTypes, SDValue &Mask) {
17111 if (ResVT != RetTy || N0->getOpcode() != ISD::VSELECT ||
17113 return SDValue();
17114 Mask = N0->getOperand(0);
17115 SDValue Ext = N0->getOperand(1);
17116 if (Ext->getOpcode() != ExtendCode)
17117 return SDValue();
17118 SDValue A = Ext->getOperand(0);
17119 if (ExtTypeMatches(A, ExtTypes))
17120 return ExtendIfNeeded(A, ExtendCode);
17121 return SDValue();
17122 };
17123 auto IsVMLAV = [&](MVT RetTy, unsigned ExtendCode, ArrayRef<MVT> ExtTypes,
17124 SDValue &A, SDValue &B) {
17125 // For a vmla we are trying to match a larger pattern:
17126 // ExtA = sext/zext A
17127 // ExtB = sext/zext B
17128 // Mul = mul ExtA, ExtB
17129 // vecreduce.add Mul
17130 // There might also be en extra extend between the mul and the addreduce, so
17131 // long as the bitwidth is high enough to make them equivalent (for example
17132 // original v8i16 might be mul at v8i32 and the reduce happens at v8i64).
17133 if (ResVT != RetTy)
17134 return false;
17135 SDValue Mul = N0;
17136 if (Mul->getOpcode() == ExtendCode &&
17137 Mul->getOperand(0).getScalarValueSizeInBits() * 2 >=
17138 ResVT.getScalarSizeInBits())
17139 Mul = Mul->getOperand(0);
17140 if (Mul->getOpcode() != ISD::MUL)
17141 return false;
17142 SDValue ExtA = Mul->getOperand(0);
17143 SDValue ExtB = Mul->getOperand(1);
17144 if (ExtA->getOpcode() != ExtendCode || ExtB->getOpcode() != ExtendCode)
17145 return false;
17146 A = ExtA->getOperand(0);
17147 B = ExtB->getOperand(0);
17148 if (ExtTypeMatches(A, ExtTypes) && ExtTypeMatches(B, ExtTypes)) {
17149 A = ExtendIfNeeded(A, ExtendCode);
17150 B = ExtendIfNeeded(B, ExtendCode);
17151 return true;
17152 }
17153 return false;
17154 };
17155 auto IsPredVMLAV = [&](MVT RetTy, unsigned ExtendCode, ArrayRef<MVT> ExtTypes,
17156 SDValue &A, SDValue &B, SDValue &Mask) {
17157 // Same as the pattern above with a select for the zero predicated lanes
17158 // ExtA = sext/zext A
17159 // ExtB = sext/zext B
17160 // Mul = mul ExtA, ExtB
17161 // N0 = select Mask, Mul, 0
17162 // vecreduce.add N0
17163 if (ResVT != RetTy || N0->getOpcode() != ISD::VSELECT ||
17165 return false;
17166 Mask = N0->getOperand(0);
17167 SDValue Mul = N0->getOperand(1);
17168 if (Mul->getOpcode() == ExtendCode &&
17169 Mul->getOperand(0).getScalarValueSizeInBits() * 2 >=
17170 ResVT.getScalarSizeInBits())
17171 Mul = Mul->getOperand(0);
17172 if (Mul->getOpcode() != ISD::MUL)
17173 return false;
17174 SDValue ExtA = Mul->getOperand(0);
17175 SDValue ExtB = Mul->getOperand(1);
17176 if (ExtA->getOpcode() != ExtendCode || ExtB->getOpcode() != ExtendCode)
17177 return false;
17178 A = ExtA->getOperand(0);
17179 B = ExtB->getOperand(0);
17180 if (ExtTypeMatches(A, ExtTypes) && ExtTypeMatches(B, ExtTypes)) {
17181 A = ExtendIfNeeded(A, ExtendCode);
17182 B = ExtendIfNeeded(B, ExtendCode);
17183 return true;
17184 }
17185 return false;
17186 };
17187 auto Create64bitNode = [&](unsigned Opcode, ArrayRef<SDValue> Ops) {
17188 // Split illegal MVT::v16i8->i64 vector reductions into two legal v8i16->i64
17189 // reductions. The operands are extended with MVEEXT, but as they are
17190 // reductions the lane orders do not matter. MVEEXT may be combined with
17191 // loads to produce two extending loads, or else they will be expanded to
17192 // VREV/VMOVL.
17193 EVT VT = Ops[0].getValueType();
17194 if (VT == MVT::v16i8) {
17195 assert((Opcode == ARMISD::VMLALVs || Opcode == ARMISD::VMLALVu) &&
17196 "Unexpected illegal long reduction opcode");
17197 bool IsUnsigned = Opcode == ARMISD::VMLALVu;
17198
17199 SDValue Ext0 =
17200 DAG.getNode(IsUnsigned ? ARMISD::MVEZEXT : ARMISD::MVESEXT, dl,
17201 DAG.getVTList(MVT::v8i16, MVT::v8i16), Ops[0]);
17202 SDValue Ext1 =
17203 DAG.getNode(IsUnsigned ? ARMISD::MVEZEXT : ARMISD::MVESEXT, dl,
17204 DAG.getVTList(MVT::v8i16, MVT::v8i16), Ops[1]);
17205
17206 SDValue MLA0 = DAG.getNode(Opcode, dl, DAG.getVTList(MVT::i32, MVT::i32),
17207 Ext0, Ext1);
17208 SDValue MLA1 =
17209 DAG.getNode(IsUnsigned ? ARMISD::VMLALVAu : ARMISD::VMLALVAs, dl,
17210 DAG.getVTList(MVT::i32, MVT::i32), MLA0, MLA0.getValue(1),
17211 Ext0.getValue(1), Ext1.getValue(1));
17212 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, MLA1, MLA1.getValue(1));
17213 }
17214 SDValue Node = DAG.getNode(Opcode, dl, {MVT::i32, MVT::i32}, Ops);
17215 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, Node,
17216 SDValue(Node.getNode(), 1));
17217 };
17218
17219 SDValue A, B;
17220 SDValue Mask;
17221 if (IsVMLAV(MVT::i32, ISD::SIGN_EXTEND, {MVT::v8i16, MVT::v16i8}, A, B))
17222 return DAG.getNode(ARMISD::VMLAVs, dl, ResVT, A, B);
17223 if (IsVMLAV(MVT::i32, ISD::ZERO_EXTEND, {MVT::v8i16, MVT::v16i8}, A, B))
17224 return DAG.getNode(ARMISD::VMLAVu, dl, ResVT, A, B);
17225 if (IsVMLAV(MVT::i64, ISD::SIGN_EXTEND, {MVT::v16i8, MVT::v8i16, MVT::v4i32},
17226 A, B))
17227 return Create64bitNode(ARMISD::VMLALVs, {A, B});
17228 if (IsVMLAV(MVT::i64, ISD::ZERO_EXTEND, {MVT::v16i8, MVT::v8i16, MVT::v4i32},
17229 A, B))
17230 return Create64bitNode(ARMISD::VMLALVu, {A, B});
17231 if (IsVMLAV(MVT::i16, ISD::SIGN_EXTEND, {MVT::v16i8}, A, B))
17232 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17233 DAG.getNode(ARMISD::VMLAVs, dl, MVT::i32, A, B));
17234 if (IsVMLAV(MVT::i16, ISD::ZERO_EXTEND, {MVT::v16i8}, A, B))
17235 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17236 DAG.getNode(ARMISD::VMLAVu, dl, MVT::i32, A, B));
17237
17238 if (IsPredVMLAV(MVT::i32, ISD::SIGN_EXTEND, {MVT::v8i16, MVT::v16i8}, A, B,
17239 Mask))
17240 return DAG.getNode(ARMISD::VMLAVps, dl, ResVT, A, B, Mask);
17241 if (IsPredVMLAV(MVT::i32, ISD::ZERO_EXTEND, {MVT::v8i16, MVT::v16i8}, A, B,
17242 Mask))
17243 return DAG.getNode(ARMISD::VMLAVpu, dl, ResVT, A, B, Mask);
17244 if (IsPredVMLAV(MVT::i64, ISD::SIGN_EXTEND, {MVT::v8i16, MVT::v4i32}, A, B,
17245 Mask))
17246 return Create64bitNode(ARMISD::VMLALVps, {A, B, Mask});
17247 if (IsPredVMLAV(MVT::i64, ISD::ZERO_EXTEND, {MVT::v8i16, MVT::v4i32}, A, B,
17248 Mask))
17249 return Create64bitNode(ARMISD::VMLALVpu, {A, B, Mask});
17250 if (IsPredVMLAV(MVT::i16, ISD::SIGN_EXTEND, {MVT::v16i8}, A, B, Mask))
17251 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17252 DAG.getNode(ARMISD::VMLAVps, dl, MVT::i32, A, B, Mask));
17253 if (IsPredVMLAV(MVT::i16, ISD::ZERO_EXTEND, {MVT::v16i8}, A, B, Mask))
17254 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17255 DAG.getNode(ARMISD::VMLAVpu, dl, MVT::i32, A, B, Mask));
17256
17257 if (SDValue A = IsVADDV(MVT::i32, ISD::SIGN_EXTEND, {MVT::v8i16, MVT::v16i8}))
17258 return DAG.getNode(ARMISD::VADDVs, dl, ResVT, A);
17259 if (SDValue A = IsVADDV(MVT::i32, ISD::ZERO_EXTEND, {MVT::v8i16, MVT::v16i8}))
17260 return DAG.getNode(ARMISD::VADDVu, dl, ResVT, A);
17261 if (SDValue A = IsVADDV(MVT::i64, ISD::SIGN_EXTEND, {MVT::v4i32}))
17262 return Create64bitNode(ARMISD::VADDLVs, {A});
17263 if (SDValue A = IsVADDV(MVT::i64, ISD::ZERO_EXTEND, {MVT::v4i32}))
17264 return Create64bitNode(ARMISD::VADDLVu, {A});
17265 if (SDValue A = IsVADDV(MVT::i16, ISD::SIGN_EXTEND, {MVT::v16i8}))
17266 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17267 DAG.getNode(ARMISD::VADDVs, dl, MVT::i32, A));
17268 if (SDValue A = IsVADDV(MVT::i16, ISD::ZERO_EXTEND, {MVT::v16i8}))
17269 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17270 DAG.getNode(ARMISD::VADDVu, dl, MVT::i32, A));
17271
17272 if (SDValue A = IsPredVADDV(MVT::i32, ISD::SIGN_EXTEND, {MVT::v8i16, MVT::v16i8}, Mask))
17273 return DAG.getNode(ARMISD::VADDVps, dl, ResVT, A, Mask);
17274 if (SDValue A = IsPredVADDV(MVT::i32, ISD::ZERO_EXTEND, {MVT::v8i16, MVT::v16i8}, Mask))
17275 return DAG.getNode(ARMISD::VADDVpu, dl, ResVT, A, Mask);
17276 if (SDValue A = IsPredVADDV(MVT::i64, ISD::SIGN_EXTEND, {MVT::v4i32}, Mask))
17277 return Create64bitNode(ARMISD::VADDLVps, {A, Mask});
17278 if (SDValue A = IsPredVADDV(MVT::i64, ISD::ZERO_EXTEND, {MVT::v4i32}, Mask))
17279 return Create64bitNode(ARMISD::VADDLVpu, {A, Mask});
17280 if (SDValue A = IsPredVADDV(MVT::i16, ISD::SIGN_EXTEND, {MVT::v16i8}, Mask))
17281 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17282 DAG.getNode(ARMISD::VADDVps, dl, MVT::i32, A, Mask));
17283 if (SDValue A = IsPredVADDV(MVT::i16, ISD::ZERO_EXTEND, {MVT::v16i8}, Mask))
17284 return DAG.getNode(ISD::TRUNCATE, dl, ResVT,
17285 DAG.getNode(ARMISD::VADDVpu, dl, MVT::i32, A, Mask));
17286
17287 // Some complications. We can get a case where the two inputs of the mul are
17288 // the same, then the output sext will have been helpfully converted to a
17289 // zext. Turn it back.
17290 SDValue Op = N0;
17291 if (Op->getOpcode() == ISD::VSELECT)
17292 Op = Op->getOperand(1);
17293 if (Op->getOpcode() == ISD::ZERO_EXTEND &&
17294 Op->getOperand(0)->getOpcode() == ISD::MUL) {
17295 SDValue Mul = Op->getOperand(0);
17296 if (Mul->getOperand(0) == Mul->getOperand(1) &&
17297 Mul->getOperand(0)->getOpcode() == ISD::SIGN_EXTEND) {
17298 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND, dl, N0->getValueType(0), Mul);
17299 if (Op != N0)
17300 Ext = DAG.getNode(ISD::VSELECT, dl, N0->getValueType(0),
17301 N0->getOperand(0), Ext, N0->getOperand(2));
17302 return DAG.getNode(ISD::VECREDUCE_ADD, dl, ResVT, Ext);
17303 }
17304 }
17305
17306 return SDValue();
17307}
17308
17309// Looks for vaddv(shuffle) or vmlav(shuffle, shuffle), with a shuffle where all
17310// the lanes are used. Due to the reduction being commutative the shuffle can be
17311// removed.
17313 unsigned VecOp = N->getOperand(0).getValueType().isVector() ? 0 : 2;
17314 auto *Shuf = dyn_cast<ShuffleVectorSDNode>(N->getOperand(VecOp));
17315 if (!Shuf || !Shuf->getOperand(1).isUndef())
17316 return SDValue();
17317
17318 // Check all elements are used once in the mask.
17319 ArrayRef<int> Mask = Shuf->getMask();
17320 APInt SetElts(Mask.size(), 0);
17321 for (int E : Mask) {
17322 if (E < 0 || E >= (int)Mask.size())
17323 return SDValue();
17324 SetElts.setBit(E);
17325 }
17326 if (!SetElts.isAllOnes())
17327 return SDValue();
17328
17329 if (N->getNumOperands() != VecOp + 1) {
17330 auto *Shuf2 = dyn_cast<ShuffleVectorSDNode>(N->getOperand(VecOp + 1));
17331 if (!Shuf2 || !Shuf2->getOperand(1).isUndef() || Shuf2->getMask() != Mask)
17332 return SDValue();
17333 }
17334
17336 for (SDValue Op : N->ops()) {
17337 if (Op.getValueType().isVector())
17338 Ops.push_back(Op.getOperand(0));
17339 else
17340 Ops.push_back(Op);
17341 }
17342 return DAG.getNode(N->getOpcode(), SDLoc(N), N->getVTList(), Ops);
17343}
17344
17347 SDValue Op0 = N->getOperand(0);
17348 SDValue Op1 = N->getOperand(1);
17349 unsigned IsTop = N->getConstantOperandVal(2);
17350
17351 // VMOVNT a undef -> a
17352 // VMOVNB a undef -> a
17353 // VMOVNB undef a -> a
17354 if (Op1->isUndef())
17355 return Op0;
17356 if (Op0->isUndef() && !IsTop)
17357 return Op1;
17358
17359 // VMOVNt(c, VQMOVNb(a, b)) => VQMOVNt(c, b)
17360 // VMOVNb(c, VQMOVNb(a, b)) => VQMOVNb(c, b)
17361 if ((Op1->getOpcode() == ARMISD::VQMOVNs ||
17362 Op1->getOpcode() == ARMISD::VQMOVNu) &&
17363 Op1->getConstantOperandVal(2) == 0)
17364 return DCI.DAG.getNode(Op1->getOpcode(), SDLoc(Op1), N->getValueType(0),
17365 Op0, Op1->getOperand(1), N->getOperand(2));
17366
17367 // Only the bottom lanes from Qm (Op1) and either the top or bottom lanes from
17368 // Qd (Op0) are demanded from a VMOVN, depending on whether we are inserting
17369 // into the top or bottom lanes.
17370 unsigned NumElts = N->getValueType(0).getVectorNumElements();
17371 APInt Op1DemandedElts = APInt::getSplat(NumElts, APInt::getLowBitsSet(2, 1));
17372 APInt Op0DemandedElts =
17373 IsTop ? Op1DemandedElts
17374 : APInt::getSplat(NumElts, APInt::getHighBitsSet(2, 1));
17375
17376 const TargetLowering &TLI = DCI.DAG.getTargetLoweringInfo();
17377 if (TLI.SimplifyDemandedVectorElts(Op0, Op0DemandedElts, DCI))
17378 return SDValue(N, 0);
17379 if (TLI.SimplifyDemandedVectorElts(Op1, Op1DemandedElts, DCI))
17380 return SDValue(N, 0);
17381
17382 return SDValue();
17383}
17384
17387 SDValue Op0 = N->getOperand(0);
17388 unsigned IsTop = N->getConstantOperandVal(2);
17389
17390 unsigned NumElts = N->getValueType(0).getVectorNumElements();
17391 APInt Op0DemandedElts =
17392 APInt::getSplat(NumElts, IsTop ? APInt::getLowBitsSet(2, 1)
17393 : APInt::getHighBitsSet(2, 1));
17394
17395 const TargetLowering &TLI = DCI.DAG.getTargetLoweringInfo();
17396 if (TLI.SimplifyDemandedVectorElts(Op0, Op0DemandedElts, DCI))
17397 return SDValue(N, 0);
17398 return SDValue();
17399}
17400
17403 EVT VT = N->getValueType(0);
17404 SDValue LHS = N->getOperand(0);
17405 SDValue RHS = N->getOperand(1);
17406
17407 auto *Shuf0 = dyn_cast<ShuffleVectorSDNode>(LHS);
17408 auto *Shuf1 = dyn_cast<ShuffleVectorSDNode>(RHS);
17409 // Turn VQDMULH(shuffle, shuffle) -> shuffle(VQDMULH)
17410 if (Shuf0 && Shuf1 && Shuf0->getMask().equals(Shuf1->getMask()) &&
17411 LHS.getOperand(1).isUndef() && RHS.getOperand(1).isUndef() &&
17412 (LHS.hasOneUse() || RHS.hasOneUse() || LHS == RHS)) {
17413 SDLoc DL(N);
17414 SDValue NewBinOp = DCI.DAG.getNode(N->getOpcode(), DL, VT,
17415 LHS.getOperand(0), RHS.getOperand(0));
17416 SDValue UndefV = LHS.getOperand(1);
17417 return DCI.DAG.getVectorShuffle(VT, DL, NewBinOp, UndefV, Shuf0->getMask());
17418 }
17419 return SDValue();
17420}
17421
17423 SDLoc DL(N);
17424 SDValue Op0 = N->getOperand(0);
17425 SDValue Op1 = N->getOperand(1);
17426
17427 // Turn X << -C -> X >> C and viceversa. The negative shifts can come up from
17428 // uses of the intrinsics.
17429 if (auto C = dyn_cast<ConstantSDNode>(N->getOperand(2))) {
17430 int ShiftAmt = C->getSExtValue();
17431 if (ShiftAmt == 0) {
17432 SDValue Merge = DAG.getMergeValues({Op0, Op1}, DL);
17433 DAG.ReplaceAllUsesWith(N, Merge.getNode());
17434 return SDValue();
17435 }
17436
17437 if (ShiftAmt >= -32 && ShiftAmt < 0) {
17438 unsigned NewOpcode =
17439 N->getOpcode() == ARMISD::LSLL ? ARMISD::LSRL : ARMISD::LSLL;
17440 SDValue NewShift = DAG.getNode(NewOpcode, DL, N->getVTList(), Op0, Op1,
17441 DAG.getConstant(-ShiftAmt, DL, MVT::i32));
17442 DAG.ReplaceAllUsesWith(N, NewShift.getNode());
17443 return NewShift;
17444 }
17445 }
17446
17447 return SDValue();
17448}
17449
17450/// PerformIntrinsicCombine - ARM-specific DAG combining for intrinsics.
17452 DAGCombinerInfo &DCI) const {
17453 SelectionDAG &DAG = DCI.DAG;
17454 unsigned IntNo = N->getConstantOperandVal(0);
17455 switch (IntNo) {
17456 default:
17457 // Don't do anything for most intrinsics.
17458 break;
17459
17460 // Vector shifts: check for immediate versions and lower them.
17461 // Note: This is done during DAG combining instead of DAG legalizing because
17462 // the build_vectors for 64-bit vector element shift counts are generally
17463 // not legal, and it is hard to see their values after they get legalized to
17464 // loads from a constant pool.
17465 case Intrinsic::arm_neon_vshifts:
17466 case Intrinsic::arm_neon_vshiftu:
17467 case Intrinsic::arm_neon_vrshifts:
17468 case Intrinsic::arm_neon_vrshiftu:
17469 case Intrinsic::arm_neon_vrshiftn:
17470 case Intrinsic::arm_neon_vqshifts:
17471 case Intrinsic::arm_neon_vqshiftu:
17472 case Intrinsic::arm_neon_vqshiftsu:
17473 case Intrinsic::arm_neon_vqshiftns:
17474 case Intrinsic::arm_neon_vqshiftnu:
17475 case Intrinsic::arm_neon_vqshiftnsu:
17476 case Intrinsic::arm_neon_vqrshiftns:
17477 case Intrinsic::arm_neon_vqrshiftnu:
17478 case Intrinsic::arm_neon_vqrshiftnsu: {
17479 EVT VT = N->getOperand(1).getValueType();
17480 int64_t Cnt;
17481 unsigned VShiftOpc = 0;
17482
17483 switch (IntNo) {
17484 case Intrinsic::arm_neon_vshifts:
17485 case Intrinsic::arm_neon_vshiftu:
17486 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt)) {
17487 VShiftOpc = ARMISD::VSHLIMM;
17488 break;
17489 }
17490 if (isVShiftRImm(N->getOperand(2), VT, false, true, Cnt)) {
17491 VShiftOpc = (IntNo == Intrinsic::arm_neon_vshifts ? ARMISD::VSHRsIMM
17492 : ARMISD::VSHRuIMM);
17493 break;
17494 }
17495 return SDValue();
17496
17497 case Intrinsic::arm_neon_vrshifts:
17498 case Intrinsic::arm_neon_vrshiftu:
17499 if (isVShiftRImm(N->getOperand(2), VT, false, true, Cnt))
17500 break;
17501 return SDValue();
17502
17503 case Intrinsic::arm_neon_vqshifts:
17504 case Intrinsic::arm_neon_vqshiftu:
17505 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt))
17506 break;
17507 return SDValue();
17508
17509 case Intrinsic::arm_neon_vqshiftsu:
17510 if (isVShiftLImm(N->getOperand(2), VT, false, Cnt))
17511 break;
17512 llvm_unreachable("invalid shift count for vqshlu intrinsic");
17513
17514 case Intrinsic::arm_neon_vrshiftn:
17515 case Intrinsic::arm_neon_vqshiftns:
17516 case Intrinsic::arm_neon_vqshiftnu:
17517 case Intrinsic::arm_neon_vqshiftnsu:
17518 case Intrinsic::arm_neon_vqrshiftns:
17519 case Intrinsic::arm_neon_vqrshiftnu:
17520 case Intrinsic::arm_neon_vqrshiftnsu:
17521 // Narrowing shifts require an immediate right shift.
17522 if (isVShiftRImm(N->getOperand(2), VT, true, true, Cnt))
17523 break;
17524 llvm_unreachable("invalid shift count for narrowing vector shift "
17525 "intrinsic");
17526
17527 default:
17528 llvm_unreachable("unhandled vector shift");
17529 }
17530
17531 switch (IntNo) {
17532 case Intrinsic::arm_neon_vshifts:
17533 case Intrinsic::arm_neon_vshiftu:
17534 // Opcode already set above.
17535 break;
17536 case Intrinsic::arm_neon_vrshifts:
17537 VShiftOpc = ARMISD::VRSHRsIMM;
17538 break;
17539 case Intrinsic::arm_neon_vrshiftu:
17540 VShiftOpc = ARMISD::VRSHRuIMM;
17541 break;
17542 case Intrinsic::arm_neon_vrshiftn:
17543 VShiftOpc = ARMISD::VRSHRNIMM;
17544 break;
17545 case Intrinsic::arm_neon_vqshifts:
17546 VShiftOpc = ARMISD::VQSHLsIMM;
17547 break;
17548 case Intrinsic::arm_neon_vqshiftu:
17549 VShiftOpc = ARMISD::VQSHLuIMM;
17550 break;
17551 case Intrinsic::arm_neon_vqshiftsu:
17552 VShiftOpc = ARMISD::VQSHLsuIMM;
17553 break;
17554 case Intrinsic::arm_neon_vqshiftns:
17555 VShiftOpc = ARMISD::VQSHRNsIMM;
17556 break;
17557 case Intrinsic::arm_neon_vqshiftnu:
17558 VShiftOpc = ARMISD::VQSHRNuIMM;
17559 break;
17560 case Intrinsic::arm_neon_vqshiftnsu:
17561 VShiftOpc = ARMISD::VQSHRNsuIMM;
17562 break;
17563 case Intrinsic::arm_neon_vqrshiftns:
17564 VShiftOpc = ARMISD::VQRSHRNsIMM;
17565 break;
17566 case Intrinsic::arm_neon_vqrshiftnu:
17567 VShiftOpc = ARMISD::VQRSHRNuIMM;
17568 break;
17569 case Intrinsic::arm_neon_vqrshiftnsu:
17570 VShiftOpc = ARMISD::VQRSHRNsuIMM;
17571 break;
17572 }
17573
17574 SDLoc dl(N);
17575 return DAG.getNode(VShiftOpc, dl, N->getValueType(0),
17576 N->getOperand(1), DAG.getConstant(Cnt, dl, MVT::i32));
17577 }
17578
17579 case Intrinsic::arm_neon_vshiftins: {
17580 EVT VT = N->getOperand(1).getValueType();
17581 int64_t Cnt;
17582 unsigned VShiftOpc = 0;
17583
17584 if (isVShiftLImm(N->getOperand(3), VT, false, Cnt))
17585 VShiftOpc = ARMISD::VSLIIMM;
17586 else if (isVShiftRImm(N->getOperand(3), VT, false, true, Cnt))
17587 VShiftOpc = ARMISD::VSRIIMM;
17588 else {
17589 llvm_unreachable("invalid shift count for vsli/vsri intrinsic");
17590 }
17591
17592 SDLoc dl(N);
17593 return DAG.getNode(VShiftOpc, dl, N->getValueType(0),
17594 N->getOperand(1), N->getOperand(2),
17595 DAG.getConstant(Cnt, dl, MVT::i32));
17596 }
17597
17598 case Intrinsic::arm_neon_vqrshifts:
17599 case Intrinsic::arm_neon_vqrshiftu:
17600 // No immediate versions of these to check for.
17601 break;
17602
17603 case Intrinsic::arm_neon_vbsl: {
17604 SDLoc dl(N);
17605 return DAG.getNode(ARMISD::VBSP, dl, N->getValueType(0), N->getOperand(1),
17606 N->getOperand(2), N->getOperand(3));
17607 }
17608 case Intrinsic::arm_mve_vqdmlah:
17609 case Intrinsic::arm_mve_vqdmlash:
17610 case Intrinsic::arm_mve_vqrdmlah:
17611 case Intrinsic::arm_mve_vqrdmlash:
17612 case Intrinsic::arm_mve_vmla_n_predicated:
17613 case Intrinsic::arm_mve_vmlas_n_predicated:
17614 case Intrinsic::arm_mve_vqdmlah_predicated:
17615 case Intrinsic::arm_mve_vqdmlash_predicated:
17616 case Intrinsic::arm_mve_vqrdmlah_predicated:
17617 case Intrinsic::arm_mve_vqrdmlash_predicated: {
17618 // These intrinsics all take an i32 scalar operand which is narrowed to the
17619 // size of a single lane of the vector type they return. So we don't need
17620 // any bits of that operand above that point, which allows us to eliminate
17621 // uxth/sxth.
17622 unsigned BitWidth = N->getValueType(0).getScalarSizeInBits();
17623 APInt DemandedMask = APInt::getLowBitsSet(32, BitWidth);
17624 if (SimplifyDemandedBits(N->getOperand(3), DemandedMask, DCI))
17625 return SDValue();
17626 break;
17627 }
17628
17629 case Intrinsic::arm_mve_minv:
17630 case Intrinsic::arm_mve_maxv:
17631 case Intrinsic::arm_mve_minav:
17632 case Intrinsic::arm_mve_maxav:
17633 case Intrinsic::arm_mve_minv_predicated:
17634 case Intrinsic::arm_mve_maxv_predicated:
17635 case Intrinsic::arm_mve_minav_predicated:
17636 case Intrinsic::arm_mve_maxav_predicated: {
17637 // These intrinsics all take an i32 scalar operand which is narrowed to the
17638 // size of a single lane of the vector type they take as the other input.
17639 unsigned BitWidth = N->getOperand(2)->getValueType(0).getScalarSizeInBits();
17640 APInt DemandedMask = APInt::getLowBitsSet(32, BitWidth);
17641 if (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))
17642 return SDValue();
17643 break;
17644 }
17645
17646 case Intrinsic::arm_mve_addv: {
17647 // Turn this intrinsic straight into the appropriate ARMISD::VADDV node,
17648 // which allow PerformADDVecReduce to turn it into VADDLV when possible.
17649 bool Unsigned = N->getConstantOperandVal(2);
17650 unsigned Opc = Unsigned ? ARMISD::VADDVu : ARMISD::VADDVs;
17651 return DAG.getNode(Opc, SDLoc(N), N->getVTList(), N->getOperand(1));
17652 }
17653
17654 case Intrinsic::arm_mve_addlv:
17655 case Intrinsic::arm_mve_addlv_predicated: {
17656 // Same for these, but ARMISD::VADDLV has to be followed by a BUILD_PAIR
17657 // which recombines the two outputs into an i64
17658 bool Unsigned = N->getConstantOperandVal(2);
17659 unsigned Opc = IntNo == Intrinsic::arm_mve_addlv ?
17660 (Unsigned ? ARMISD::VADDLVu : ARMISD::VADDLVs) :
17661 (Unsigned ? ARMISD::VADDLVpu : ARMISD::VADDLVps);
17662
17664 for (unsigned i = 1, e = N->getNumOperands(); i < e; i++)
17665 if (i != 2) // skip the unsigned flag
17666 Ops.push_back(N->getOperand(i));
17667
17668 SDLoc dl(N);
17669 SDValue val = DAG.getNode(Opc, dl, {MVT::i32, MVT::i32}, Ops);
17670 return DAG.getNode(ISD::BUILD_PAIR, dl, MVT::i64, val.getValue(0),
17671 val.getValue(1));
17672 }
17673 }
17674
17675 return SDValue();
17676}
17677
17679 EVT VT = Y.getValueType();
17680 if (!VT.isVector())
17681 return hasAndNotCompare(Y);
17682 if (Subtarget->hasMVEIntegerOps())
17683 return VT.is128BitVector();
17684 if (Subtarget->hasNEON())
17685 return VT.is64BitVector() || VT.is128BitVector();
17686 return false;
17687}
17688
17689/// PerformShiftCombine - Checks for immediate versions of vector shifts and
17690/// lowers them. As with the vector shift intrinsics, this is done during DAG
17691/// combining instead of DAG legalizing because the build_vectors for 64-bit
17692/// vector element shift counts are generally not legal, and it is hard to see
17693/// their values after they get legalized to loads from a constant pool.
17696 const ARMSubtarget *ST) {
17697 SelectionDAG &DAG = DCI.DAG;
17698 EVT VT = N->getValueType(0);
17699
17700 if (ST->isThumb1Only() && N->getOpcode() == ISD::SHL && VT == MVT::i32 &&
17701 N->getOperand(0)->getOpcode() == ISD::AND &&
17702 N->getOperand(0)->hasOneUse()) {
17703 if (DCI.isBeforeLegalize() || DCI.isCalledByLegalizer())
17704 return SDValue();
17705 // Look for the pattern (shl (and x, AndMask), ShiftAmt). This doesn't
17706 // usually show up because instcombine prefers to canonicalize it to
17707 // (and (shl x, ShiftAmt) (shl AndMask, ShiftAmt)), but the shift can come
17708 // out of GEP lowering in some cases.
17709 SDValue N0 = N->getOperand(0);
17710 ConstantSDNode *ShiftAmtNode = dyn_cast<ConstantSDNode>(N->getOperand(1));
17711 if (!ShiftAmtNode)
17712 return SDValue();
17713 uint32_t ShiftAmt = static_cast<uint32_t>(ShiftAmtNode->getZExtValue());
17714 ConstantSDNode *AndMaskNode = dyn_cast<ConstantSDNode>(N0->getOperand(1));
17715 if (!AndMaskNode)
17716 return SDValue();
17717 uint32_t AndMask = static_cast<uint32_t>(AndMaskNode->getZExtValue());
17718 // Don't transform uxtb/uxth.
17719 if (AndMask == 255 || AndMask == 65535)
17720 return SDValue();
17721 if (isMask_32(AndMask)) {
17722 uint32_t MaskedBits = llvm::countl_zero(AndMask);
17723 if (MaskedBits > ShiftAmt) {
17724 SDLoc DL(N);
17725 SDValue SHL = DAG.getNode(ISD::SHL, DL, MVT::i32, N0->getOperand(0),
17726 DAG.getConstant(MaskedBits, DL, MVT::i32));
17727 return DAG.getNode(
17728 ISD::SRL, DL, MVT::i32, SHL,
17729 DAG.getConstant(MaskedBits - ShiftAmt, DL, MVT::i32));
17730 }
17731 }
17732 }
17733
17734 // Nothing to be done for scalar shifts.
17735 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
17736 if (!VT.isVector() || !TLI.isTypeLegal(VT))
17737 return SDValue();
17738 if (ST->hasMVEIntegerOps())
17739 return SDValue();
17740
17741 int64_t Cnt;
17742
17743 switch (N->getOpcode()) {
17744 default: llvm_unreachable("unexpected shift opcode");
17745
17746 case ISD::SHL:
17747 if (isVShiftLImm(N->getOperand(1), VT, false, Cnt)) {
17748 SDLoc dl(N);
17749 return DAG.getNode(ARMISD::VSHLIMM, dl, VT, N->getOperand(0),
17750 DAG.getConstant(Cnt, dl, MVT::i32));
17751 }
17752 break;
17753
17754 case ISD::SRA:
17755 case ISD::SRL:
17756 if (isVShiftRImm(N->getOperand(1), VT, false, false, Cnt)) {
17757 unsigned VShiftOpc =
17758 (N->getOpcode() == ISD::SRA ? ARMISD::VSHRsIMM : ARMISD::VSHRuIMM);
17759 SDLoc dl(N);
17760 return DAG.getNode(VShiftOpc, dl, VT, N->getOperand(0),
17761 DAG.getConstant(Cnt, dl, MVT::i32));
17762 }
17763 }
17764 return SDValue();
17765}
17766
17767// Look for a sign/zero/fpextend extend of a larger than legal load. This can be
17768// split into multiple extending loads, which are simpler to deal with than an
17769// arbitrary extend. For fp extends we use an integer extending load and a VCVTL
17770// to convert the type to an f32.
17772 SDValue N0 = N->getOperand(0);
17773 if (N0.getOpcode() != ISD::LOAD)
17774 return SDValue();
17776 if (!LD->isSimple() || !N0.hasOneUse() || LD->isIndexed() ||
17777 LD->getExtensionType() != ISD::NON_EXTLOAD)
17778 return SDValue();
17779 EVT FromVT = LD->getValueType(0);
17780 EVT ToVT = N->getValueType(0);
17781 if (!ToVT.isVector())
17782 return SDValue();
17784 EVT ToEltVT = ToVT.getVectorElementType();
17785 EVT FromEltVT = FromVT.getVectorElementType();
17786
17787 unsigned NumElements = 0;
17788 if (ToEltVT == MVT::i32 && FromEltVT == MVT::i8)
17789 NumElements = 4;
17790 if (ToEltVT == MVT::f32 && FromEltVT == MVT::f16)
17791 NumElements = 4;
17792 if (NumElements == 0 ||
17793 (FromEltVT != MVT::f16 && FromVT.getVectorNumElements() == NumElements) ||
17794 FromVT.getVectorNumElements() % NumElements != 0 ||
17795 !isPowerOf2_32(NumElements))
17796 return SDValue();
17797
17798 LLVMContext &C = *DAG.getContext();
17799 SDLoc DL(LD);
17800 // Details about the old load
17801 SDValue Ch = LD->getChain();
17802 SDValue BasePtr = LD->getBasePtr();
17803 Align Alignment = LD->getBaseAlign();
17804 MachineMemOperand::Flags MMOFlags = LD->getMemOperand()->getFlags();
17805 AAMDNodes AAInfo = LD->getAAInfo();
17806
17807 ISD::LoadExtType NewExtType =
17808 N->getOpcode() == ISD::SIGN_EXTEND ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
17809 SDValue Offset = DAG.getPOISON(BasePtr.getValueType());
17810 EVT NewFromVT = EVT::getVectorVT(
17811 C, EVT::getIntegerVT(C, FromEltVT.getScalarSizeInBits()), NumElements);
17812 EVT NewToVT = EVT::getVectorVT(
17813 C, EVT::getIntegerVT(C, ToEltVT.getScalarSizeInBits()), NumElements);
17814
17817 for (unsigned i = 0; i < FromVT.getVectorNumElements() / NumElements; i++) {
17818 unsigned NewOffset = (i * NewFromVT.getSizeInBits()) / 8;
17819 SDValue NewPtr =
17820 DAG.getObjectPtrOffset(DL, BasePtr, TypeSize::getFixed(NewOffset));
17821
17822 SDValue NewLoad =
17823 DAG.getLoad(ISD::UNINDEXED, NewExtType, NewToVT, DL, Ch, NewPtr, Offset,
17824 LD->getPointerInfo().getWithOffset(NewOffset), NewFromVT,
17825 Alignment, MMOFlags, AAInfo);
17826 Loads.push_back(NewLoad);
17827 Chains.push_back(SDValue(NewLoad.getNode(), 1));
17828 }
17829
17830 // Float truncs need to extended with VCVTB's into their floating point types.
17831 if (FromEltVT == MVT::f16) {
17833
17834 for (unsigned i = 0; i < Loads.size(); i++) {
17835 SDValue LoadBC =
17836 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, MVT::v8f16, Loads[i]);
17837 SDValue FPExt = DAG.getNode(ARMISD::VCVTL, DL, MVT::v4f32, LoadBC,
17838 DAG.getConstant(0, DL, MVT::i32));
17839 Extends.push_back(FPExt);
17840 }
17841
17842 Loads = Extends;
17843 }
17844
17845 SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
17846 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), NewChain);
17847 return DAG.getNode(ISD::CONCAT_VECTORS, DL, ToVT, Loads);
17848}
17849
17850/// PerformExtendCombine - Target-specific DAG combining for ISD::SIGN_EXTEND,
17851/// ISD::ZERO_EXTEND, and ISD::ANY_EXTEND.
17853 const ARMSubtarget *ST) {
17854 SDValue N0 = N->getOperand(0);
17855 EVT VT = N->getValueType(0);
17856 SDLoc DL(N);
17857
17858 // Check for sign- and zero-extensions of vector extract operations of 8- and
17859 // 16-bit vector elements. NEON and MVE support these directly. They are
17860 // handled during DAG combining because type legalization will promote them
17861 // to 32-bit types and it is messy to recognize the operations after that.
17862 if ((ST->hasNEON() || ST->hasMVEIntegerOps()) &&
17864 SDValue Vec = N0.getOperand(0);
17865 SDValue Lane = N0.getOperand(1);
17866 EVT EltVT = N0.getValueType();
17867 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
17868
17869 if (VT == MVT::i32 &&
17870 (EltVT == MVT::i8 || EltVT == MVT::i16) &&
17871 TLI.isTypeLegal(Vec.getValueType()) &&
17872 isa<ConstantSDNode>(Lane)) {
17873
17874 unsigned Opc = 0;
17875 switch (N->getOpcode()) {
17876 default: llvm_unreachable("unexpected opcode");
17877 case ISD::SIGN_EXTEND:
17878 Opc = ARMISD::VGETLANEs;
17879 break;
17880 case ISD::ZERO_EXTEND:
17881 case ISD::ANY_EXTEND:
17882 Opc = ARMISD::VGETLANEu;
17883 break;
17884 }
17885 return DAG.getNode(Opc, DL, VT, Vec, Lane);
17886 }
17887 }
17888
17889 if (ST->hasMVEIntegerOps())
17890 if (SDValue NewLoad = PerformSplittingToWideningLoad(N, DAG))
17891 return NewLoad;
17892
17893 // Combine sext(buildvector(..)) to buildvector(sext(..)) to help avoid
17894 // difficult to lower i1 buildvector.
17895 if (ST->hasMVEIntegerOps() && N0.getValueType().getScalarSizeInBits() == 1 &&
17896 N0.getOpcode() == ISD::BUILD_VECTOR && VT.getScalarSizeInBits() <= 32) {
17898 for (unsigned I = 0; I < N0.getNumOperands(); I++) {
17899 SDValue InReg = N0.getOperand(I);
17900 if (N->getOpcode() == ISD::ZERO_EXTEND)
17901 InReg = DAG.getNode(ISD::AND, DL, InReg.getValueType(), InReg,
17902 DAG.getConstant(1, DL, InReg.getValueType()));
17903 else if (N->getOpcode() == ISD::SIGN_EXTEND)
17904 InReg = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, InReg.getValueType(),
17905 InReg, DAG.getValueType(MVT::i1));
17906 SDValue Ext = DAG.getNode(N->getOpcode(), DL, MVT::i32, InReg);
17907 Ops.push_back(Ext);
17908 }
17909 return DAG.getNode(ISD::BUILD_VECTOR, DL, VT, Ops);
17910 }
17911
17912 return SDValue();
17913}
17914
17916 const ARMSubtarget *ST) {
17917 if (ST->hasMVEFloatOps())
17918 if (SDValue NewLoad = PerformSplittingToWideningLoad(N, DAG))
17919 return NewLoad;
17920
17921 return SDValue();
17922}
17923
17924// Lower smin(smax(x, C1), C2) to ssat or usat, if they have saturating
17925// constant bounds.
17927 const ARMSubtarget *Subtarget) {
17928 if ((Subtarget->isThumb() || !Subtarget->hasV6Ops()) &&
17929 !Subtarget->isThumb2())
17930 return SDValue();
17931
17932 EVT VT = Op.getValueType();
17933 SDValue Op0 = Op.getOperand(0);
17934
17935 if (VT != MVT::i32 ||
17936 (Op0.getOpcode() != ISD::SMIN && Op0.getOpcode() != ISD::SMAX) ||
17937 !isa<ConstantSDNode>(Op.getOperand(1)) ||
17939 return SDValue();
17940
17941 SDValue Min = Op;
17942 SDValue Max = Op0;
17943 SDValue Input = Op0.getOperand(0);
17944 if (Min.getOpcode() == ISD::SMAX)
17945 std::swap(Min, Max);
17946
17947 if (Min.getOpcode() != ISD::SMIN || Max.getOpcode() != ISD::SMAX)
17948 return SDValue();
17949
17950 APInt MinC = Min.getConstantOperandAPInt(1);
17951 APInt MaxC = Max.getConstantOperandAPInt(1);
17952 if (MaxC.sgt(MinC))
17953 return SDValue();
17954
17955 SDLoc DL(Op);
17956
17957 // A clamp whose bounds are already a saturation range maps to a single
17958 // SSAT / USAT.
17959 if ((MinC + 1).isPowerOf2()) {
17960 if (MinC == ~MaxC)
17961 return DAG.getNode(ARMISD::SSAT, DL, VT, Input,
17962 DAG.getConstant(MinC.countr_one(), DL, VT));
17963 if (MaxC == 0)
17964 return DAG.getNode(ARMISD::USAT, DL, VT, Input,
17965 DAG.getConstant(MinC.countr_one(), DL, VT));
17966 }
17967
17968 // For power-of-two clamp widths, convert the range to be zero-centered,
17969 // apply SSAT, and convert the result back.
17970 //
17971 // Width = Hi - Lo + 1
17972 // Center = Lo + Width / 2
17973 // Result = ssat(X - Center) + Center
17974 //
17975 // The idea is to shift the input so that the clamp range is centered
17976 // around zero, apply ssat, and then shift the result back.
17977 //
17978 // For example clamp(X, -118, 137) -> Width = 256, Center = 10, so it becomes
17979 // ssat(X - 10, 8) + 10
17980
17981 APInt Width = MinC - MaxC + 1;
17982 if (!Width.isPowerOf2() || Width.isOne())
17983 return SDValue();
17984 unsigned SatBit = Width.logBase2() - 1; // ssat to SatBit + 1 signed bits
17985 APInt Center = MaxC + Width.lshr(1);
17986
17987 // The rewrite is only valid when X - Center does not overflow;
17988 SDValue NegC = DAG.getConstant(-Center, DL, VT);
17990 return SDValue();
17991
17992 SDValue Shifted = DAG.getNode(ISD::ADD, DL, VT, Input, NegC);
17993 SDValue Sat = DAG.getNode(ARMISD::SSAT, DL, VT, Shifted,
17994 DAG.getConstant(SatBit, DL, VT));
17995 return DAG.getNode(ISD::ADD, DL, VT, Sat, DAG.getConstant(Center, DL, VT));
17996}
17997
17998/// PerformMinMaxCombine - Target-specific DAG combining for creating truncating
17999/// saturates.
18001 const ARMSubtarget *ST) {
18002 EVT VT = N->getValueType(0);
18003 SDValue N0 = N->getOperand(0);
18004
18005 if (VT == MVT::i32)
18006 return PerformMinMaxToSatCombine(SDValue(N, 0), DAG, ST);
18007
18008 if (!ST->hasMVEIntegerOps())
18009 return SDValue();
18010
18011 if (SDValue V = PerformVQDMULHCombine(N, DAG))
18012 return V;
18013
18014 if (VT != MVT::v4i32 && VT != MVT::v8i16)
18015 return SDValue();
18016
18017 auto IsSignedSaturate = [&](SDNode *Min, SDNode *Max) {
18018 // Check one is a smin and the other is a smax
18019 if (Min->getOpcode() != ISD::SMIN)
18020 std::swap(Min, Max);
18021 if (Min->getOpcode() != ISD::SMIN || Max->getOpcode() != ISD::SMAX)
18022 return false;
18023
18024 APInt SaturateC;
18025 if (VT == MVT::v4i32)
18026 SaturateC = APInt(32, (1 << 15) - 1, true);
18027 else //if (VT == MVT::v8i16)
18028 SaturateC = APInt(16, (1 << 7) - 1, true);
18029
18030 APInt MinC, MaxC;
18031 if (!ISD::isConstantSplatVector(Min->getOperand(1).getNode(), MinC) ||
18032 MinC != SaturateC)
18033 return false;
18034 if (!ISD::isConstantSplatVector(Max->getOperand(1).getNode(), MaxC) ||
18035 MaxC != ~SaturateC)
18036 return false;
18037 return true;
18038 };
18039
18040 if (IsSignedSaturate(N, N0.getNode())) {
18041 SDLoc DL(N);
18042 MVT ExtVT, HalfVT;
18043 if (VT == MVT::v4i32) {
18044 HalfVT = MVT::v8i16;
18045 ExtVT = MVT::v4i16;
18046 } else { // if (VT == MVT::v8i16)
18047 HalfVT = MVT::v16i8;
18048 ExtVT = MVT::v8i8;
18049 }
18050
18051 // Create a VQMOVNB with undef top lanes, then signed extended into the top
18052 // half. That extend will hopefully be removed if only the bottom bits are
18053 // demanded (though a truncating store, for example).
18054 SDValue VQMOVN =
18055 DAG.getNode(ARMISD::VQMOVNs, DL, HalfVT, DAG.getUNDEF(HalfVT),
18056 N0->getOperand(0), DAG.getConstant(0, DL, MVT::i32));
18057 SDValue Bitcast = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, VQMOVN);
18058 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, Bitcast,
18059 DAG.getValueType(ExtVT));
18060 }
18061
18062 auto IsUnsignedSaturate = [&](SDNode *Min) {
18063 // For unsigned, we just need to check for <= 0xffff
18064 if (Min->getOpcode() != ISD::UMIN)
18065 return false;
18066
18067 APInt SaturateC;
18068 if (VT == MVT::v4i32)
18069 SaturateC = APInt(32, (1 << 16) - 1, true);
18070 else //if (VT == MVT::v8i16)
18071 SaturateC = APInt(16, (1 << 8) - 1, true);
18072
18073 APInt MinC;
18074 if (!ISD::isConstantSplatVector(Min->getOperand(1).getNode(), MinC) ||
18075 MinC != SaturateC)
18076 return false;
18077 return true;
18078 };
18079
18080 if (IsUnsignedSaturate(N)) {
18081 SDLoc DL(N);
18082 MVT HalfVT;
18083 unsigned ExtConst;
18084 if (VT == MVT::v4i32) {
18085 HalfVT = MVT::v8i16;
18086 ExtConst = 0x0000FFFF;
18087 } else { //if (VT == MVT::v8i16)
18088 HalfVT = MVT::v16i8;
18089 ExtConst = 0x00FF;
18090 }
18091
18092 // Create a VQMOVNB with undef top lanes, then ZExt into the top half with
18093 // an AND. That extend will hopefully be removed if only the bottom bits are
18094 // demanded (though a truncating store, for example).
18095 SDValue VQMOVN =
18096 DAG.getNode(ARMISD::VQMOVNu, DL, HalfVT, DAG.getUNDEF(HalfVT), N0,
18097 DAG.getConstant(0, DL, MVT::i32));
18098 SDValue Bitcast = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, VQMOVN);
18099 return DAG.getNode(ISD::AND, DL, VT, Bitcast,
18100 DAG.getConstant(ExtConst, DL, VT));
18101 }
18102
18103 return SDValue();
18104}
18105
18108 if (!C)
18109 return nullptr;
18110 const APInt *CV = &C->getAPIntValue();
18111 return CV->isPowerOf2() ? CV : nullptr;
18112}
18113
18115 // If we have a CMOV, OR and AND combination such as:
18116 // if (x & CN)
18117 // y |= CM;
18118 //
18119 // And:
18120 // * CN is a single bit;
18121 // * All bits covered by CM are known zero in y
18122 //
18123 // Then we can convert this into a sequence of BFI instructions. This will
18124 // always be a win if CM is a single bit, will always be no worse than the
18125 // TST&OR sequence if CM is two bits, and for thumb will be no worse if CM is
18126 // three bits (due to the extra IT instruction).
18127
18128 SDValue Op0 = CMOV->getOperand(0);
18129 SDValue Op1 = CMOV->getOperand(1);
18130 auto CC = CMOV->getConstantOperandAPInt(2).getLimitedValue();
18131 SDValue CmpZ = CMOV->getOperand(3);
18132
18133 // The compare must be against zero.
18134 if (!isNullConstant(CmpZ->getOperand(1)))
18135 return SDValue();
18136
18137 assert(CmpZ->getOpcode() == ARMISD::CMPZ);
18138 SDValue And = CmpZ->getOperand(0);
18139 if (And->getOpcode() != ISD::AND)
18140 return SDValue();
18141 const APInt *AndC = isPowerOf2Constant(And->getOperand(1));
18142 if (!AndC)
18143 return SDValue();
18144 SDValue X = And->getOperand(0);
18145
18146 if (CC == ARMCC::EQ) {
18147 // We're performing an "equal to zero" compare. Swap the operands so we
18148 // canonicalize on a "not equal to zero" compare.
18149 std::swap(Op0, Op1);
18150 } else {
18151 assert(CC == ARMCC::NE && "How can a CMPZ node not be EQ or NE?");
18152 }
18153
18154 if (Op1->getOpcode() != ISD::OR)
18155 return SDValue();
18156
18158 if (!OrC)
18159 return SDValue();
18160 SDValue Y = Op1->getOperand(0);
18161
18162 if (Op0 != Y)
18163 return SDValue();
18164
18165 // Now, is it profitable to continue?
18166 APInt OrCI = OrC->getAPIntValue();
18167 unsigned Heuristic = Subtarget->isThumb() ? 3 : 2;
18168 if (OrCI.popcount() > Heuristic)
18169 return SDValue();
18170
18171 // Lastly, can we determine that the bits defined by OrCI
18172 // are zero in Y?
18174 if ((OrCI & Known.Zero) != OrCI)
18175 return SDValue();
18176
18177 // OK, we can do the combine.
18178 SDValue V = Y;
18179 SDLoc dl(X);
18180 EVT VT = X.getValueType();
18181 unsigned BitInX = AndC->logBase2();
18182
18183 if (BitInX != 0) {
18184 // We must shift X first.
18185 X = DAG.getNode(ISD::SRL, dl, VT, X,
18186 DAG.getConstant(BitInX, dl, VT));
18187 }
18188
18189 for (unsigned BitInY = 0, NumActiveBits = OrCI.getActiveBits();
18190 BitInY < NumActiveBits; ++BitInY) {
18191 if (OrCI[BitInY] == 0)
18192 continue;
18193 APInt Mask(VT.getSizeInBits(), 0);
18194 Mask.setBit(BitInY);
18195 V = DAG.getNode(ARMISD::BFI, dl, VT, V, X,
18196 // Confusingly, the operand is an *inverted* mask.
18197 DAG.getConstant(~Mask, dl, VT));
18198 }
18199
18200 return V;
18201}
18202
18203// Given N, the value controlling the conditional branch, search for the loop
18204// intrinsic, returning it, along with how the value is used. We need to handle
18205// patterns such as the following:
18206// (brcond (xor (setcc (loop.decrement), 0, ne), 1), exit)
18207// (brcond (setcc (loop.decrement), 0, eq), exit)
18208// (brcond (setcc (loop.decrement), 0, ne), header)
18210 bool &Negate) {
18211 switch (N->getOpcode()) {
18212 default:
18213 break;
18214 case ISD::XOR: {
18215 if (!isa<ConstantSDNode>(N.getOperand(1)))
18216 return SDValue();
18217 if (!cast<ConstantSDNode>(N.getOperand(1))->isOne())
18218 return SDValue();
18219 Negate = !Negate;
18220 return SearchLoopIntrinsic(N.getOperand(0), CC, Imm, Negate);
18221 }
18222 case ISD::SETCC: {
18223 auto *Const = dyn_cast<ConstantSDNode>(N.getOperand(1));
18224 if (!Const)
18225 return SDValue();
18226 if (Const->isZero())
18227 Imm = 0;
18228 else if (Const->isOne())
18229 Imm = 1;
18230 else
18231 return SDValue();
18232 CC = cast<CondCodeSDNode>(N.getOperand(2))->get();
18233 return SearchLoopIntrinsic(N->getOperand(0), CC, Imm, Negate);
18234 }
18236 unsigned IntOp = N.getConstantOperandVal(1);
18237 if (IntOp != Intrinsic::test_start_loop_iterations &&
18238 IntOp != Intrinsic::loop_decrement_reg)
18239 return SDValue();
18240 return N;
18241 }
18242 }
18243 return SDValue();
18244}
18245
18248 const ARMSubtarget *ST) {
18249
18250 // The hwloop intrinsics that we're interested are used for control-flow,
18251 // either for entering or exiting the loop:
18252 // - test.start.loop.iterations will test whether its operand is zero. If it
18253 // is zero, the proceeding branch should not enter the loop.
18254 // - loop.decrement.reg also tests whether its operand is zero. If it is
18255 // zero, the proceeding branch should not branch back to the beginning of
18256 // the loop.
18257 // So here, we need to check that how the brcond is using the result of each
18258 // of the intrinsics to ensure that we're branching to the right place at the
18259 // right time.
18260
18261 ISD::CondCode CC;
18262 SDValue Cond;
18263 int Imm = 1;
18264 bool Negate = false;
18265 SDValue Chain = N->getOperand(0);
18266 SDValue Dest;
18267
18268 if (N->getOpcode() == ISD::BRCOND) {
18269 CC = ISD::SETEQ;
18270 Cond = N->getOperand(1);
18271 Dest = N->getOperand(2);
18272 } else {
18273 assert(N->getOpcode() == ISD::BR_CC && "Expected BRCOND or BR_CC!");
18274 CC = cast<CondCodeSDNode>(N->getOperand(1))->get();
18275 Cond = N->getOperand(2);
18276 Dest = N->getOperand(4);
18277 if (auto *Const = dyn_cast<ConstantSDNode>(N->getOperand(3))) {
18278 if (!Const->isOne() && !Const->isZero())
18279 return SDValue();
18280 Imm = Const->getZExtValue();
18281 } else
18282 return SDValue();
18283 }
18284
18285 SDValue Int = SearchLoopIntrinsic(Cond, CC, Imm, Negate);
18286 if (!Int)
18287 return SDValue();
18288
18289 if (Negate)
18290 CC = ISD::getSetCCInverse(CC, /* Integer inverse */ MVT::i32);
18291
18292 auto IsTrueIfZero = [](ISD::CondCode CC, int Imm) {
18293 return (CC == ISD::SETEQ && Imm == 0) ||
18294 (CC == ISD::SETNE && Imm == 1) ||
18295 (CC == ISD::SETLT && Imm == 1) ||
18296 (CC == ISD::SETULT && Imm == 1);
18297 };
18298
18299 auto IsFalseIfZero = [](ISD::CondCode CC, int Imm) {
18300 return (CC == ISD::SETEQ && Imm == 1) ||
18301 (CC == ISD::SETNE && Imm == 0) ||
18302 (CC == ISD::SETGT && Imm == 0) ||
18303 (CC == ISD::SETUGT && Imm == 0) ||
18304 (CC == ISD::SETGE && Imm == 1) ||
18305 (CC == ISD::SETUGE && Imm == 1);
18306 };
18307
18308 assert((IsTrueIfZero(CC, Imm) || IsFalseIfZero(CC, Imm)) &&
18309 "unsupported condition");
18310
18311 SDLoc dl(Int);
18312 SelectionDAG &DAG = DCI.DAG;
18313 SDValue Elements = Int.getOperand(2);
18314 unsigned IntOp = Int->getConstantOperandVal(1);
18315 assert((N->hasOneUse() && N->user_begin()->getOpcode() == ISD::BR) &&
18316 "expected single br user");
18317 SDNode *Br = *N->user_begin();
18318 SDValue OtherTarget = Br->getOperand(1);
18319
18320 // Update the unconditional branch to branch to the given Dest.
18321 auto UpdateUncondBr = [](SDNode *Br, SDValue Dest, SelectionDAG &DAG) {
18322 SDValue NewBrOps[] = { Br->getOperand(0), Dest };
18323 SDValue NewBr = DAG.getNode(ISD::BR, SDLoc(Br), MVT::Other, NewBrOps);
18324 DAG.ReplaceAllUsesOfValueWith(SDValue(Br, 0), NewBr);
18325 };
18326
18327 if (IntOp == Intrinsic::test_start_loop_iterations) {
18328 SDValue Res;
18329 SDValue Setup = DAG.getNode(ARMISD::WLSSETUP, dl, MVT::i32, Elements);
18330 // We expect this 'instruction' to branch when the counter is zero.
18331 if (IsTrueIfZero(CC, Imm)) {
18332 SDValue Ops[] = {Chain, Setup, Dest};
18333 Res = DAG.getNode(ARMISD::WLS, dl, MVT::Other, Ops);
18334 } else {
18335 // The logic is the reverse of what we need for WLS, so find the other
18336 // basic block target: the target of the proceeding br.
18337 UpdateUncondBr(Br, Dest, DAG);
18338
18339 SDValue Ops[] = {Chain, Setup, OtherTarget};
18340 Res = DAG.getNode(ARMISD::WLS, dl, MVT::Other, Ops);
18341 }
18342 // Update LR count to the new value
18343 DAG.ReplaceAllUsesOfValueWith(Int.getValue(0), Setup);
18344 // Update chain
18345 DAG.ReplaceAllUsesOfValueWith(Int.getValue(2), Int.getOperand(0));
18346 return Res;
18347 } else {
18348 SDValue Size =
18349 DAG.getTargetConstant(Int.getConstantOperandVal(3), dl, MVT::i32);
18350 SDValue Args[] = { Int.getOperand(0), Elements, Size, };
18351 SDValue LoopDec = DAG.getNode(ARMISD::LOOP_DEC, dl,
18352 DAG.getVTList(MVT::i32, MVT::Other), Args);
18353 DAG.ReplaceAllUsesWith(Int.getNode(), LoopDec.getNode());
18354
18355 // We expect this instruction to branch when the count is not zero.
18356 SDValue Target = IsFalseIfZero(CC, Imm) ? Dest : OtherTarget;
18357
18358 // Update the unconditional branch to target the loop preheader if we've
18359 // found the condition has been reversed.
18360 if (Target == OtherTarget)
18361 UpdateUncondBr(Br, Dest, DAG);
18362
18363 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other,
18364 SDValue(LoopDec.getNode(), 1), Chain);
18365
18366 SDValue EndArgs[] = { Chain, SDValue(LoopDec.getNode(), 0), Target };
18367 return DAG.getNode(ARMISD::LE, dl, MVT::Other, EndArgs);
18368 }
18369 return SDValue();
18370}
18371
18372/// PerformBRCONDCombine - Target-specific DAG combining for ARMISD::BRCOND.
18373SDValue
18375 SDValue Cmp = N->getOperand(3);
18376 if (Cmp.getOpcode() != ARMISD::CMPZ)
18377 // Only looking at NE cases.
18378 return SDValue();
18379
18380 SDLoc dl(N);
18381 SDValue LHS = Cmp.getOperand(0);
18382 SDValue RHS = Cmp.getOperand(1);
18383 SDValue Chain = N->getOperand(0);
18384 SDValue BB = N->getOperand(1);
18385 SDValue ARMcc = N->getOperand(2);
18387
18388 // (brcond Chain BB ne (cmpz (and (cmov 0 1 CC Flags) 1) 0))
18389 // -> (brcond Chain BB CC Flags)
18390 if (CC == ARMCC::NE && LHS.getOpcode() == ISD::AND && LHS->hasOneUse() &&
18391 LHS->getOperand(0)->getOpcode() == ARMISD::CMOV &&
18392 LHS->getOperand(0)->hasOneUse() &&
18393 isNullConstant(LHS->getOperand(0)->getOperand(0)) &&
18394 isOneConstant(LHS->getOperand(0)->getOperand(1)) &&
18395 isOneConstant(LHS->getOperand(1)) && isNullConstant(RHS)) {
18396 return DAG.getNode(ARMISD::BRCOND, dl, MVT::Other, Chain, BB,
18397 LHS->getOperand(0)->getOperand(2),
18398 LHS->getOperand(0)->getOperand(3));
18399 }
18400
18401 return SDValue();
18402}
18403
18404/// PerformCMOVCombine - Target-specific DAG combining for ARMISD::CMOV.
18405SDValue
18407 SDLoc dl(N);
18408 EVT VT = N->getValueType(0);
18409 SDValue FalseVal = N->getOperand(0);
18410 SDValue TrueVal = N->getOperand(1);
18411 SDValue ARMcc = N->getOperand(2);
18412 SDValue Cmp = N->getOperand(3);
18413
18414 // Try to form CSINV etc.
18415 unsigned Opcode;
18416 bool InvertCond;
18417 if (SDValue CSetOp =
18418 matchCSET(Opcode, InvertCond, TrueVal, FalseVal, Subtarget)) {
18419 if (InvertCond) {
18420 ARMCC::CondCodes CondCode =
18421 (ARMCC::CondCodes)cast<const ConstantSDNode>(ARMcc)->getZExtValue();
18422 CondCode = ARMCC::getOppositeCondition(CondCode);
18423 ARMcc = DAG.getConstant(CondCode, SDLoc(ARMcc), MVT::i32);
18424 }
18425 return DAG.getNode(Opcode, dl, VT, CSetOp, CSetOp, ARMcc, Cmp);
18426 }
18427
18428 if (Cmp.getOpcode() != ARMISD::CMPZ)
18429 // Only looking at EQ and NE cases.
18430 return SDValue();
18431
18432 SDValue LHS = Cmp.getOperand(0);
18433 SDValue RHS = Cmp.getOperand(1);
18435
18436 // BFI is only available on V6T2+.
18437 if (!Subtarget->isThumb1Only() && Subtarget->hasV6T2Ops()) {
18439 if (R)
18440 return R;
18441 }
18442
18443 // Simplify
18444 // mov r1, r0
18445 // cmp r1, x
18446 // mov r0, y
18447 // moveq r0, x
18448 // to
18449 // cmp r0, x
18450 // movne r0, y
18451 //
18452 // mov r1, r0
18453 // cmp r1, x
18454 // mov r0, x
18455 // movne r0, y
18456 // to
18457 // cmp r0, x
18458 // movne r0, y
18459 /// FIXME: Turn this into a target neutral optimization?
18460 SDValue Res;
18461 if (CC == ARMCC::NE && FalseVal == RHS && FalseVal != LHS) {
18462 Res = DAG.getNode(ARMISD::CMOV, dl, VT, LHS, TrueVal, ARMcc, Cmp);
18463 } else if (CC == ARMCC::EQ && TrueVal == RHS) {
18464 SDValue ARMcc;
18465 SDValue NewCmp = getARMCmp(LHS, RHS, ISD::SETNE, ARMcc, DAG, dl);
18466 Res = DAG.getNode(ARMISD::CMOV, dl, VT, LHS, FalseVal, ARMcc, NewCmp);
18467 }
18468
18469 // (cmov F T ne (cmpz (cmov 0 1 CC Flags) 0))
18470 // -> (cmov F T CC Flags)
18471 if (CC == ARMCC::NE && LHS.getOpcode() == ARMISD::CMOV && LHS->hasOneUse() &&
18472 isNullConstant(LHS->getOperand(0)) && isOneConstant(LHS->getOperand(1)) &&
18473 isNullConstant(RHS)) {
18474 return DAG.getNode(ARMISD::CMOV, dl, VT, FalseVal, TrueVal,
18475 LHS->getOperand(2), LHS->getOperand(3));
18476 }
18477
18478 if (!VT.isInteger())
18479 return SDValue();
18480
18481 // Fold away an unnecessary CMPZ/CMOV
18482 // CMOV A, B, C1, (CMPZ (CMOV 1, 0, C2, D), 0) ->
18483 // if C1==EQ -> CMOV A, B, C2, D
18484 // if C1==NE -> CMOV A, B, NOT(C2), D
18485 if (N->getConstantOperandVal(2) == ARMCC::EQ ||
18486 N->getConstantOperandVal(2) == ARMCC::NE) {
18488 if (SDValue C = IsCMPZCSINC(N->getOperand(3).getNode(), Cond)) {
18489 if (N->getConstantOperandVal(2) == ARMCC::NE)
18491 return DAG.getNode(N->getOpcode(), SDLoc(N), MVT::i32, N->getOperand(0),
18492 N->getOperand(1),
18493 DAG.getConstant(Cond, SDLoc(N), MVT::i32), C);
18494 }
18495 }
18496
18497 // Materialize a boolean comparison for integers so we can avoid branching.
18498 if (isNullConstant(FalseVal)) {
18499 if (CC == ARMCC::EQ && isOneConstant(TrueVal)) {
18500 if (!Subtarget->isThumb1Only() && Subtarget->hasV5TOps()) {
18501 // If x == y then x - y == 0 and ARM's CLZ will return 32, shifting it
18502 // right 5 bits will make that 32 be 1, otherwise it will be 0.
18503 // CMOV 0, 1, ==, (CMPZ x, y) -> SRL (CTLZ (SUB x, y)), 5
18504 SDValue Sub = DAG.getNode(ISD::SUB, dl, VT, LHS, RHS);
18505 Res = DAG.getNode(ISD::SRL, dl, VT, DAG.getNode(ISD::CTLZ, dl, VT, Sub),
18506 DAG.getConstant(5, dl, MVT::i32));
18507 } else {
18508 // CMOV 0, 1, ==, (CMPZ x, y) ->
18509 // (UADDO_CARRY (SUB x, y), t:0, t:1)
18510 // where t = (USUBO_CARRY 0, (SUB x, y), 0)
18511 //
18512 // The USUBO_CARRY computes 0 - (x - y) and this will give a borrow when
18513 // x != y. In other words, a carry C == 1 when x == y, C == 0
18514 // otherwise.
18515 // The final UADDO_CARRY computes
18516 // x - y + (0 - (x - y)) + C == C
18517 SDValue Sub = DAG.getNode(ISD::SUB, dl, VT, LHS, RHS);
18518 SDVTList VTs = DAG.getVTList(VT, MVT::i32);
18519 SDValue Neg = DAG.getNode(ISD::USUBO, dl, VTs, FalseVal, Sub);
18520 // ISD::USUBO_CARRY returns a borrow but we want the carry here
18521 // actually.
18522 SDValue Carry =
18523 DAG.getNode(ISD::SUB, dl, MVT::i32,
18524 DAG.getConstant(1, dl, MVT::i32), Neg.getValue(1));
18525 Res = DAG.getNode(ISD::UADDO_CARRY, dl, VTs, Sub, Neg, Carry);
18526 }
18527 } else if (CC == ARMCC::NE && !isNullConstant(RHS) &&
18528 (!Subtarget->isThumb1Only() || isPowerOf2Constant(TrueVal))) {
18529 // This seems pointless but will allow us to combine it further below.
18530 // CMOV 0, z, !=, (CMPZ x, y) -> CMOV (SUBC x, y), z, !=, (SUBC x, y):1
18531 SDValue Sub =
18532 DAG.getNode(ARMISD::SUBC, dl, DAG.getVTList(VT, MVT::i32), LHS, RHS);
18533 Res = DAG.getNode(ARMISD::CMOV, dl, VT, Sub, TrueVal, ARMcc,
18534 Sub.getValue(1));
18535 FalseVal = Sub;
18536 }
18537 } else if (isNullConstant(TrueVal)) {
18538 if (CC == ARMCC::EQ && !isNullConstant(RHS) &&
18539 (!Subtarget->isThumb1Only() || isPowerOf2Constant(FalseVal))) {
18540 // This seems pointless but will allow us to combine it further below
18541 // Note that we change == for != as this is the dual for the case above.
18542 // CMOV z, 0, ==, (CMPZ x, y) -> CMOV (SUBC x, y), z, !=, (SUBC x, y):1
18543 SDValue Sub =
18544 DAG.getNode(ARMISD::SUBC, dl, DAG.getVTList(VT, MVT::i32), LHS, RHS);
18545 Res = DAG.getNode(ARMISD::CMOV, dl, VT, Sub, FalseVal,
18546 DAG.getConstant(ARMCC::NE, dl, MVT::i32),
18547 Sub.getValue(1));
18548 FalseVal = Sub;
18549 }
18550 }
18551
18552 // On Thumb1, the DAG above may be further combined if z is a power of 2
18553 // (z == 2 ^ K).
18554 // CMOV (SUBC x, y), z, !=, (SUBC x, y):1 ->
18555 // t1 = (USUBO (SUB x, y), 1)
18556 // t2 = (USUBO_CARRY (SUB x, y), t1:0, t1:1)
18557 // Result = if K != 0 then (SHL t2:0, K) else t2:0
18558 //
18559 // This also handles the special case of comparing against zero; it's
18560 // essentially, the same pattern, except there's no SUBC:
18561 // CMOV x, z, !=, (CMPZ x, 0) ->
18562 // t1 = (USUBO x, 1)
18563 // t2 = (USUBO_CARRY x, t1:0, t1:1)
18564 // Result = if K != 0 then (SHL t2:0, K) else t2:0
18565 const APInt *TrueConst;
18566 if (Subtarget->isThumb1Only() && CC == ARMCC::NE &&
18567 ((FalseVal.getOpcode() == ARMISD::SUBC && FalseVal.getOperand(0) == LHS &&
18568 FalseVal.getOperand(1) == RHS) ||
18569 (FalseVal == LHS && isNullConstant(RHS))) &&
18570 (TrueConst = isPowerOf2Constant(TrueVal))) {
18571 SDVTList VTs = DAG.getVTList(VT, MVT::i32);
18572 unsigned ShiftAmount = TrueConst->logBase2();
18573 if (ShiftAmount)
18574 TrueVal = DAG.getConstant(1, dl, VT);
18575 SDValue Subc = DAG.getNode(ISD::USUBO, dl, VTs, FalseVal, TrueVal);
18576 Res = DAG.getNode(ISD::USUBO_CARRY, dl, VTs, FalseVal, Subc,
18577 Subc.getValue(1));
18578
18579 if (ShiftAmount)
18580 Res = DAG.getNode(ISD::SHL, dl, VT, Res,
18581 DAG.getConstant(ShiftAmount, dl, MVT::i32));
18582 }
18583
18584 if (Res.getNode()) {
18586 // Capture demanded bits information that would be otherwise lost.
18587 if (Known.Zero == 0xfffffffe)
18588 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res,
18589 DAG.getValueType(MVT::i1));
18590 else if (Known.Zero == 0xffffff00)
18591 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res,
18592 DAG.getValueType(MVT::i8));
18593 else if (Known.Zero == 0xffff0000)
18594 Res = DAG.getNode(ISD::AssertZext, dl, MVT::i32, Res,
18595 DAG.getValueType(MVT::i16));
18596 }
18597
18598 return Res;
18599}
18600
18603 const ARMSubtarget *ST) {
18604 SelectionDAG &DAG = DCI.DAG;
18605 SDValue Src = N->getOperand(0);
18606 EVT DstVT = N->getValueType(0);
18607
18608 // Convert v4f32 bitcast (v4i32 vdup (i32)) -> v4f32 vdup (i32) under MVE.
18609 if (ST->hasMVEIntegerOps() && Src.getOpcode() == ARMISD::VDUP) {
18610 EVT SrcVT = Src.getValueType();
18611 if (SrcVT.getScalarSizeInBits() == DstVT.getScalarSizeInBits())
18612 return DAG.getNode(ARMISD::VDUP, SDLoc(N), DstVT, Src.getOperand(0));
18613 }
18614
18615 // We may have a bitcast of something that has already had this bitcast
18616 // combine performed on it, so skip past any VECTOR_REG_CASTs.
18617 if (Src.getOpcode() == ARMISD::VECTOR_REG_CAST &&
18618 Src.getOperand(0).getValueType().getScalarSizeInBits() <=
18619 Src.getValueType().getScalarSizeInBits())
18620 Src = Src.getOperand(0);
18621
18622 // Bitcast from element-wise VMOV or VMVN doesn't need VREV if the VREV that
18623 // would be generated is at least the width of the element type.
18624 EVT SrcVT = Src.getValueType();
18625 if ((Src.getOpcode() == ARMISD::VMOVIMM ||
18626 Src.getOpcode() == ARMISD::VMVNIMM ||
18627 Src.getOpcode() == ARMISD::VMOVFPIMM) &&
18628 SrcVT.getScalarSizeInBits() <= DstVT.getScalarSizeInBits() &&
18629 DAG.getDataLayout().isBigEndian())
18630 return DAG.getNode(ARMISD::VECTOR_REG_CAST, SDLoc(N), DstVT, Src);
18631
18632 // bitcast(extract(x, n)); bitcast(extract(x, n+1)) -> VMOVRRD x
18633 if (SDValue R = PerformExtractEltToVMOVRRD(N, DCI))
18634 return R;
18635
18636 return SDValue();
18637}
18638
18639// Some combines for the MVETrunc truncations legalizer helper. Also lowers the
18640// node into stack operations after legalizeOps.
18643 SelectionDAG &DAG = DCI.DAG;
18644 EVT VT = N->getValueType(0);
18645 SDLoc DL(N);
18646
18647 // MVETrunc(Undef, Undef) -> Undef
18648 if (all_of(N->ops(), [](SDValue Op) { return Op.isUndef(); }))
18649 return DAG.getUNDEF(VT);
18650
18651 // MVETrunc(MVETrunc a b, MVETrunc c, d) -> MVETrunc
18652 if (N->getNumOperands() == 2 &&
18653 N->getOperand(0).getOpcode() == ARMISD::MVETRUNC &&
18654 N->getOperand(1).getOpcode() == ARMISD::MVETRUNC)
18655 return DAG.getNode(ARMISD::MVETRUNC, DL, VT, N->getOperand(0).getOperand(0),
18656 N->getOperand(0).getOperand(1),
18657 N->getOperand(1).getOperand(0),
18658 N->getOperand(1).getOperand(1));
18659
18660 // MVETrunc(shuffle, shuffle) -> VMOVN
18661 if (N->getNumOperands() == 2 &&
18662 N->getOperand(0).getOpcode() == ISD::VECTOR_SHUFFLE &&
18663 N->getOperand(1).getOpcode() == ISD::VECTOR_SHUFFLE) {
18664 auto *S0 = cast<ShuffleVectorSDNode>(N->getOperand(0).getNode());
18665 auto *S1 = cast<ShuffleVectorSDNode>(N->getOperand(1).getNode());
18666
18667 if (S0->getOperand(0) == S1->getOperand(0) &&
18668 S0->getOperand(1) == S1->getOperand(1)) {
18669 // Construct complete shuffle mask
18670 SmallVector<int, 8> Mask(S0->getMask());
18671 Mask.append(S1->getMask().begin(), S1->getMask().end());
18672
18673 if (isVMOVNTruncMask(Mask, VT, false))
18674 return DAG.getNode(
18675 ARMISD::VMOVN, DL, VT,
18676 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, S0->getOperand(0)),
18677 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, S0->getOperand(1)),
18678 DAG.getConstant(1, DL, MVT::i32));
18679 if (isVMOVNTruncMask(Mask, VT, true))
18680 return DAG.getNode(
18681 ARMISD::VMOVN, DL, VT,
18682 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, S0->getOperand(1)),
18683 DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, S0->getOperand(0)),
18684 DAG.getConstant(1, DL, MVT::i32));
18685 }
18686 }
18687
18688 // For MVETrunc of a buildvector or shuffle, it can be beneficial to lower the
18689 // truncate to a buildvector to allow the generic optimisations to kick in.
18690 if (all_of(N->ops(), [](SDValue Op) {
18691 return Op.getOpcode() == ISD::BUILD_VECTOR ||
18692 Op.getOpcode() == ISD::VECTOR_SHUFFLE ||
18693 (Op.getOpcode() == ISD::BITCAST &&
18694 Op.getOperand(0).getOpcode() == ISD::BUILD_VECTOR);
18695 })) {
18696 SmallVector<SDValue, 8> Extracts;
18697 for (unsigned Op = 0; Op < N->getNumOperands(); Op++) {
18698 SDValue O = N->getOperand(Op);
18699 for (unsigned i = 0; i < O.getValueType().getVectorNumElements(); i++) {
18700 SDValue Ext = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, O,
18701 DAG.getConstant(i, DL, MVT::i32));
18702 Extracts.push_back(Ext);
18703 }
18704 }
18705 return DAG.getBuildVector(VT, DL, Extracts);
18706 }
18707
18708 // If we are late in the legalization process and nothing has optimised
18709 // the trunc to anything better, lower it to a stack store and reload,
18710 // performing the truncation whilst keeping the lanes in the correct order:
18711 // VSTRH.32 a, stack; VSTRH.32 b, stack+8; VLDRW.32 stack;
18712 if (!DCI.isAfterLegalizeDAG())
18713 return SDValue();
18714
18715 SDValue StackPtr = DAG.CreateStackTemporary(TypeSize::getFixed(16), Align(4));
18716 int SPFI = cast<FrameIndexSDNode>(StackPtr.getNode())->getIndex();
18717 int NumIns = N->getNumOperands();
18718 assert((NumIns == 2 || NumIns == 4) &&
18719 "Expected 2 or 4 inputs to an MVETrunc");
18720 EVT StoreVT = VT.getHalfNumVectorElementsVT(*DAG.getContext());
18721 if (N->getNumOperands() == 4)
18722 StoreVT = StoreVT.getHalfNumVectorElementsVT(*DAG.getContext());
18723
18724 SmallVector<SDValue> Chains;
18725 for (int I = 0; I < NumIns; I++) {
18726 SDValue Ptr = DAG.getNode(
18727 ISD::ADD, DL, StackPtr.getValueType(), StackPtr,
18728 DAG.getConstant(I * 16 / NumIns, DL, StackPtr.getValueType()));
18730 DAG.getMachineFunction(), SPFI, I * 16 / NumIns);
18731 SDValue Ch = DAG.getTruncStore(DAG.getEntryNode(), DL, N->getOperand(I),
18732 Ptr, MPI, StoreVT, Align(4));
18733 Chains.push_back(Ch);
18734 }
18735
18736 SDValue Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
18737 MachinePointerInfo MPI =
18739 return DAG.getLoad(VT, DL, Chain, StackPtr, MPI, Align(4));
18740}
18741
18742// Take a MVEEXT(load x) and split that into (extload x, extload x+8)
18744 SelectionDAG &DAG) {
18745 SDValue N0 = N->getOperand(0);
18747 if (!LD || !LD->isSimple() || !N0.hasOneUse() || LD->isIndexed())
18748 return SDValue();
18749
18750 EVT FromVT = LD->getMemoryVT();
18751 EVT ToVT = N->getValueType(0);
18752 if (!ToVT.isVector())
18753 return SDValue();
18754 assert(FromVT.getVectorNumElements() == ToVT.getVectorNumElements() * 2);
18755 EVT ToEltVT = ToVT.getVectorElementType();
18756 EVT FromEltVT = FromVT.getVectorElementType();
18757
18758 unsigned NumElements = 0;
18759 if (ToEltVT == MVT::i32 && (FromEltVT == MVT::i16 || FromEltVT == MVT::i8))
18760 NumElements = 4;
18761 if (ToEltVT == MVT::i16 && FromEltVT == MVT::i8)
18762 NumElements = 8;
18763 assert(NumElements != 0);
18764
18765 ISD::LoadExtType NewExtType =
18766 N->getOpcode() == ARMISD::MVESEXT ? ISD::SEXTLOAD : ISD::ZEXTLOAD;
18767 if (LD->getExtensionType() != ISD::NON_EXTLOAD &&
18768 LD->getExtensionType() != ISD::EXTLOAD &&
18769 LD->getExtensionType() != NewExtType)
18770 return SDValue();
18771
18772 LLVMContext &C = *DAG.getContext();
18773 SDLoc DL(LD);
18774 // Details about the old load
18775 SDValue Ch = LD->getChain();
18776 SDValue BasePtr = LD->getBasePtr();
18777 Align Alignment = LD->getBaseAlign();
18778 MachineMemOperand::Flags MMOFlags = LD->getMemOperand()->getFlags();
18779 AAMDNodes AAInfo = LD->getAAInfo();
18780
18781 SDValue Offset = DAG.getPOISON(BasePtr.getValueType());
18782 EVT NewFromVT = EVT::getVectorVT(
18783 C, EVT::getIntegerVT(C, FromEltVT.getScalarSizeInBits()), NumElements);
18784 EVT NewToVT = EVT::getVectorVT(
18785 C, EVT::getIntegerVT(C, ToEltVT.getScalarSizeInBits()), NumElements);
18786
18789 for (unsigned i = 0; i < FromVT.getVectorNumElements() / NumElements; i++) {
18790 unsigned NewOffset = (i * NewFromVT.getSizeInBits()) / 8;
18791 SDValue NewPtr =
18792 DAG.getObjectPtrOffset(DL, BasePtr, TypeSize::getFixed(NewOffset));
18793
18794 SDValue NewLoad =
18795 DAG.getLoad(ISD::UNINDEXED, NewExtType, NewToVT, DL, Ch, NewPtr, Offset,
18796 LD->getPointerInfo().getWithOffset(NewOffset), NewFromVT,
18797 Alignment, MMOFlags, AAInfo);
18798 Loads.push_back(NewLoad);
18799 Chains.push_back(SDValue(NewLoad.getNode(), 1));
18800 }
18801
18802 SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
18803 DAG.ReplaceAllUsesOfValueWith(SDValue(LD, 1), NewChain);
18804 return DAG.getMergeValues(Loads, DL);
18805}
18806
18807// Perform combines for MVEEXT. If it has not be optimized to anything better
18808// before lowering, it gets converted to stack store and extloads performing the
18809// extend whilst still keeping the same lane ordering.
18812 SelectionDAG &DAG = DCI.DAG;
18813 EVT VT = N->getValueType(0);
18814 SDLoc DL(N);
18815 assert(N->getNumValues() == 2 && "Expected MVEEXT with 2 elements");
18816 assert((VT == MVT::v4i32 || VT == MVT::v8i16) && "Unexpected MVEEXT type");
18817
18818 EVT ExtVT = N->getOperand(0).getValueType().getHalfNumVectorElementsVT(
18819 *DAG.getContext());
18820 auto Extend = [&](SDValue V) {
18821 SDValue VVT = DAG.getNode(ARMISD::VECTOR_REG_CAST, DL, VT, V);
18822 return N->getOpcode() == ARMISD::MVESEXT
18823 ? DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, VT, VVT,
18824 DAG.getValueType(ExtVT))
18825 : DAG.getZeroExtendInReg(VVT, DL, ExtVT);
18826 };
18827
18828 // MVEEXT(VDUP) -> SIGN_EXTEND_INREG(VDUP)
18829 if (N->getOperand(0).getOpcode() == ARMISD::VDUP) {
18830 SDValue Ext = Extend(N->getOperand(0));
18831 return DAG.getMergeValues({Ext, Ext}, DL);
18832 }
18833
18834 // MVEEXT(shuffle) -> SIGN_EXTEND_INREG/ZERO_EXTEND_INREG
18835 if (auto *SVN = dyn_cast<ShuffleVectorSDNode>(N->getOperand(0))) {
18836 ArrayRef<int> Mask = SVN->getMask();
18837 assert(Mask.size() == 2 * VT.getVectorNumElements());
18838 assert(Mask.size() == SVN->getValueType(0).getVectorNumElements());
18839 unsigned Rev = VT == MVT::v4i32 ? ARMISD::VREV32 : ARMISD::VREV16;
18840 SDValue Op0 = SVN->getOperand(0);
18841 SDValue Op1 = SVN->getOperand(1);
18842
18843 auto CheckInregMask = [&](int Start, int Offset) {
18844 for (int Idx = 0, E = VT.getVectorNumElements(); Idx < E; ++Idx)
18845 if (Mask[Start + Idx] >= 0 && Mask[Start + Idx] != Idx * 2 + Offset)
18846 return false;
18847 return true;
18848 };
18849 SDValue V0 = SDValue(N, 0);
18850 SDValue V1 = SDValue(N, 1);
18851 if (CheckInregMask(0, 0))
18852 V0 = Extend(Op0);
18853 else if (CheckInregMask(0, 1))
18854 V0 = Extend(DAG.getNode(Rev, DL, SVN->getValueType(0), Op0));
18855 else if (CheckInregMask(0, Mask.size()))
18856 V0 = Extend(Op1);
18857 else if (CheckInregMask(0, Mask.size() + 1))
18858 V0 = Extend(DAG.getNode(Rev, DL, SVN->getValueType(0), Op1));
18859
18860 if (CheckInregMask(VT.getVectorNumElements(), Mask.size()))
18861 V1 = Extend(Op1);
18862 else if (CheckInregMask(VT.getVectorNumElements(), Mask.size() + 1))
18863 V1 = Extend(DAG.getNode(Rev, DL, SVN->getValueType(0), Op1));
18864 else if (CheckInregMask(VT.getVectorNumElements(), 0))
18865 V1 = Extend(Op0);
18866 else if (CheckInregMask(VT.getVectorNumElements(), 1))
18867 V1 = Extend(DAG.getNode(Rev, DL, SVN->getValueType(0), Op0));
18868
18869 if (V0.getNode() != N || V1.getNode() != N)
18870 return DAG.getMergeValues({V0, V1}, DL);
18871 }
18872
18873 // MVEEXT(load) -> extload, extload
18874 if (N->getOperand(0)->getOpcode() == ISD::LOAD)
18876 return L;
18877
18878 if (!DCI.isAfterLegalizeDAG())
18879 return SDValue();
18880
18881 // Lower to a stack store and reload:
18882 // VSTRW.32 a, stack; VLDRH.32 stack; VLDRH.32 stack+8;
18883 SDValue StackPtr = DAG.CreateStackTemporary(TypeSize::getFixed(16), Align(4));
18884 int SPFI = cast<FrameIndexSDNode>(StackPtr.getNode())->getIndex();
18885 int NumOuts = N->getNumValues();
18886 assert((NumOuts == 2 || NumOuts == 4) &&
18887 "Expected 2 or 4 outputs to an MVEEXT");
18888 EVT LoadVT = N->getOperand(0).getValueType().getHalfNumVectorElementsVT(
18889 *DAG.getContext());
18890 if (N->getNumOperands() == 4)
18891 LoadVT = LoadVT.getHalfNumVectorElementsVT(*DAG.getContext());
18892
18893 MachinePointerInfo MPI =
18895 SDValue Chain = DAG.getStore(DAG.getEntryNode(), DL, N->getOperand(0),
18896 StackPtr, MPI, Align(4));
18897
18899 for (int I = 0; I < NumOuts; I++) {
18900 SDValue Ptr = DAG.getNode(
18901 ISD::ADD, DL, StackPtr.getValueType(), StackPtr,
18902 DAG.getConstant(I * 16 / NumOuts, DL, StackPtr.getValueType()));
18904 DAG.getMachineFunction(), SPFI, I * 16 / NumOuts);
18905 SDValue Load = DAG.getExtLoad(
18906 N->getOpcode() == ARMISD::MVESEXT ? ISD::SEXTLOAD : ISD::ZEXTLOAD, DL,
18907 VT, Chain, Ptr, MPI, LoadVT, Align(4));
18908 Loads.push_back(Load);
18909 }
18910
18911 return DAG.getMergeValues(Loads, DL);
18912}
18913
18915 DAGCombinerInfo &DCI) const {
18916 switch (N->getOpcode()) {
18917 default: break;
18918 case ISD::SELECT_CC:
18919 case ISD::SELECT: return PerformSELECTCombine(N, DCI, Subtarget);
18920 case ISD::VSELECT: return PerformVSELECTCombine(N, DCI, Subtarget);
18921 case ISD::SETCC: return PerformVSetCCToVCTPCombine(N, DCI, Subtarget);
18922 case ARMISD::ADDE: return PerformADDECombine(N, DCI, Subtarget);
18923 case ARMISD::UMLAL: return PerformUMLALCombine(N, DCI.DAG, Subtarget);
18924 case ISD::ADD: return PerformADDCombine(N, DCI, Subtarget);
18925 case ISD::SUB: return PerformSUBCombine(N, DCI, Subtarget);
18926 case ISD::MUL: return PerformMULCombine(N, DCI, Subtarget);
18927 case ISD::OR: return PerformORCombine(N, DCI, Subtarget);
18928 case ISD::XOR: return PerformXORCombine(N, DCI, Subtarget);
18929 case ISD::AND: return PerformANDCombine(N, DCI, Subtarget);
18930 case ISD::BRCOND:
18931 case ISD::BR_CC: return PerformHWLoopCombine(N, DCI, Subtarget);
18932 case ARMISD::ADDC:
18933 case ARMISD::SUBC: return PerformAddcSubcCombine(N, DCI, Subtarget);
18934 case ARMISD::SUBE: return PerformAddeSubeCombine(N, DCI, Subtarget);
18935 case ARMISD::BFI: return PerformBFICombine(N, DCI.DAG);
18936 case ARMISD::VMOVRRD: return PerformVMOVRRDCombine(N, DCI, Subtarget);
18937 case ARMISD::VMOVDRR: return PerformVMOVDRRCombine(N, DCI.DAG);
18938 case ARMISD::VMOVhr: return PerformVMOVhrCombine(N, DCI);
18939 case ARMISD::VMOVrh: return PerformVMOVrhCombine(N, DCI.DAG);
18940 case ISD::STORE: return PerformSTORECombine(N, DCI, Subtarget);
18941 case ISD::BUILD_VECTOR: return PerformBUILD_VECTORCombine(N, DCI, Subtarget);
18944 return PerformExtractEltCombine(N, DCI, Subtarget);
18948 case ARMISD::VDUPLANE: return PerformVDUPLANECombine(N, DCI, Subtarget);
18949 case ARMISD::VDUP: return PerformVDUPCombine(N, DCI.DAG, Subtarget);
18950 case ISD::FP_TO_SINT:
18951 case ISD::FP_TO_UINT:
18952 return PerformVCVTCombine(N, DCI.DAG, Subtarget);
18953 case ISD::FADD:
18954 return PerformFADDCombine(N, DCI.DAG, Subtarget);
18955 case ISD::FMUL:
18956 return PerformVMulVCTPCombine(N, DCI.DAG, Subtarget);
18958 return PerformIntrinsicCombine(N, DCI);
18959 case ISD::SHL:
18960 case ISD::SRA:
18961 case ISD::SRL:
18962 return PerformShiftCombine(N, DCI, Subtarget);
18963 case ISD::SIGN_EXTEND:
18964 case ISD::ZERO_EXTEND:
18965 case ISD::ANY_EXTEND:
18966 return PerformExtendCombine(N, DCI.DAG, Subtarget);
18967 case ISD::FP_EXTEND:
18968 return PerformFPExtendCombine(N, DCI.DAG, Subtarget);
18969 case ISD::SMIN:
18970 case ISD::UMIN:
18971 case ISD::SMAX:
18972 case ISD::UMAX:
18973 return PerformMinMaxCombine(N, DCI.DAG, Subtarget);
18974 case ARMISD::CMOV:
18975 return PerformCMOVCombine(N, DCI.DAG);
18976 case ARMISD::BRCOND:
18977 return PerformBRCONDCombine(N, DCI.DAG);
18978 case ARMISD::CMPZ:
18979 return PerformCMPZCombine(N, DCI.DAG);
18980 case ARMISD::CSINC:
18981 case ARMISD::CSINV:
18982 case ARMISD::CSNEG:
18983 return PerformCSETCombine(N, DCI.DAG);
18984 case ISD::LOAD:
18985 return PerformLOADCombine(N, DCI, Subtarget);
18986 case ARMISD::VLD1DUP:
18987 case ARMISD::VLD2DUP:
18988 case ARMISD::VLD3DUP:
18989 case ARMISD::VLD4DUP:
18990 return PerformVLDCombine(N, DCI);
18992 return PerformARMBUILD_VECTORCombine(N, DCI);
18993 case ISD::BITCAST:
18994 return PerformBITCASTCombine(N, DCI, Subtarget);
18995 case ARMISD::PREDICATE_CAST:
18996 return PerformPREDICATE_CASTCombine(N, DCI);
18997 case ARMISD::VECTOR_REG_CAST:
18998 return PerformVECTOR_REG_CASTCombine(N, DCI.DAG, Subtarget);
18999 case ARMISD::MVETRUNC:
19000 return PerformMVETruncCombine(N, DCI);
19001 case ARMISD::MVESEXT:
19002 case ARMISD::MVEZEXT:
19003 return PerformMVEExtCombine(N, DCI);
19004 case ARMISD::VCMP:
19005 return PerformVCMPCombine(N, DCI.DAG, Subtarget);
19006 case ISD::VECREDUCE_ADD:
19007 return PerformVECREDUCE_ADDCombine(N, DCI.DAG, Subtarget);
19008 case ARMISD::VADDVs:
19009 case ARMISD::VADDVu:
19010 case ARMISD::VADDLVs:
19011 case ARMISD::VADDLVu:
19012 case ARMISD::VADDLVAs:
19013 case ARMISD::VADDLVAu:
19014 case ARMISD::VMLAVs:
19015 case ARMISD::VMLAVu:
19016 case ARMISD::VMLALVs:
19017 case ARMISD::VMLALVu:
19018 case ARMISD::VMLALVAs:
19019 case ARMISD::VMLALVAu:
19020 return PerformReduceShuffleCombine(N, DCI.DAG);
19021 case ARMISD::VMOVN:
19022 return PerformVMOVNCombine(N, DCI);
19023 case ARMISD::VQMOVNs:
19024 case ARMISD::VQMOVNu:
19025 return PerformVQMOVNCombine(N, DCI);
19026 case ARMISD::VQDMULH:
19027 return PerformVQDMULHCombine(N, DCI);
19028 case ARMISD::ASRL:
19029 case ARMISD::LSRL:
19030 case ARMISD::LSLL:
19031 return PerformLongShiftCombine(N, DCI.DAG);
19032 case ARMISD::SMULWB: {
19033 unsigned BitWidth = N->getValueType(0).getSizeInBits();
19034 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 16);
19035 if (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))
19036 return SDValue();
19037 break;
19038 }
19039 case ARMISD::SMULWT: {
19040 unsigned BitWidth = N->getValueType(0).getSizeInBits();
19041 APInt DemandedMask = APInt::getHighBitsSet(BitWidth, 16);
19042 if (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI))
19043 return SDValue();
19044 break;
19045 }
19046 case ARMISD::SMLALBB:
19047 case ARMISD::QADD16b:
19048 case ARMISD::QSUB16b:
19049 case ARMISD::UQADD16b:
19050 case ARMISD::UQSUB16b: {
19051 unsigned BitWidth = N->getValueType(0).getSizeInBits();
19052 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 16);
19053 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) ||
19054 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI)))
19055 return SDValue();
19056 break;
19057 }
19058 case ARMISD::SMLALBT: {
19059 unsigned LowWidth = N->getOperand(0).getValueType().getSizeInBits();
19060 APInt LowMask = APInt::getLowBitsSet(LowWidth, 16);
19061 unsigned HighWidth = N->getOperand(1).getValueType().getSizeInBits();
19062 APInt HighMask = APInt::getHighBitsSet(HighWidth, 16);
19063 if ((SimplifyDemandedBits(N->getOperand(0), LowMask, DCI)) ||
19064 (SimplifyDemandedBits(N->getOperand(1), HighMask, DCI)))
19065 return SDValue();
19066 break;
19067 }
19068 case ARMISD::SMLALTB: {
19069 unsigned HighWidth = N->getOperand(0).getValueType().getSizeInBits();
19070 APInt HighMask = APInt::getHighBitsSet(HighWidth, 16);
19071 unsigned LowWidth = N->getOperand(1).getValueType().getSizeInBits();
19072 APInt LowMask = APInt::getLowBitsSet(LowWidth, 16);
19073 if ((SimplifyDemandedBits(N->getOperand(0), HighMask, DCI)) ||
19074 (SimplifyDemandedBits(N->getOperand(1), LowMask, DCI)))
19075 return SDValue();
19076 break;
19077 }
19078 case ARMISD::SMLALTT: {
19079 unsigned BitWidth = N->getValueType(0).getSizeInBits();
19080 APInt DemandedMask = APInt::getHighBitsSet(BitWidth, 16);
19081 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) ||
19082 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI)))
19083 return SDValue();
19084 break;
19085 }
19086 case ARMISD::QADD8b:
19087 case ARMISD::QSUB8b:
19088 case ARMISD::UQADD8b:
19089 case ARMISD::UQSUB8b: {
19090 unsigned BitWidth = N->getValueType(0).getSizeInBits();
19091 APInt DemandedMask = APInt::getLowBitsSet(BitWidth, 8);
19092 if ((SimplifyDemandedBits(N->getOperand(0), DemandedMask, DCI)) ||
19093 (SimplifyDemandedBits(N->getOperand(1), DemandedMask, DCI)))
19094 return SDValue();
19095 break;
19096 }
19097 case ARMISD::VBSP:
19098 if (N->getOperand(1) == N->getOperand(2))
19099 return N->getOperand(1);
19100 return SDValue();
19103 switch (N->getConstantOperandVal(1)) {
19104 case Intrinsic::arm_neon_vld1:
19105 case Intrinsic::arm_neon_vld1x2:
19106 case Intrinsic::arm_neon_vld1x3:
19107 case Intrinsic::arm_neon_vld1x4:
19108 case Intrinsic::arm_neon_vld2:
19109 case Intrinsic::arm_neon_vld3:
19110 case Intrinsic::arm_neon_vld4:
19111 case Intrinsic::arm_neon_vld2lane:
19112 case Intrinsic::arm_neon_vld3lane:
19113 case Intrinsic::arm_neon_vld4lane:
19114 case Intrinsic::arm_neon_vld2dup:
19115 case Intrinsic::arm_neon_vld3dup:
19116 case Intrinsic::arm_neon_vld4dup:
19117 case Intrinsic::arm_neon_vst1:
19118 case Intrinsic::arm_neon_vst1x2:
19119 case Intrinsic::arm_neon_vst1x3:
19120 case Intrinsic::arm_neon_vst1x4:
19121 case Intrinsic::arm_neon_vst2:
19122 case Intrinsic::arm_neon_vst3:
19123 case Intrinsic::arm_neon_vst4:
19124 case Intrinsic::arm_neon_vst2lane:
19125 case Intrinsic::arm_neon_vst3lane:
19126 case Intrinsic::arm_neon_vst4lane:
19127 return PerformVLDCombine(N, DCI);
19128 case Intrinsic::arm_mve_vld2q:
19129 case Intrinsic::arm_mve_vld4q:
19130 case Intrinsic::arm_mve_vst2q:
19131 case Intrinsic::arm_mve_vst4q:
19132 return PerformMVEVLDCombine(N, DCI);
19133 default: break;
19134 }
19135 break;
19136 }
19137 return SDValue();
19138}
19139
19141 EVT VT) const {
19142 return (VT == MVT::f32) && (Opc == ISD::LOAD || Opc == ISD::STORE);
19143}
19144
19146 Align Alignment,
19148 unsigned *Fast) const {
19149 // Depends what it gets converted into if the type is weird.
19150 if (!VT.isSimple())
19151 return false;
19152
19153 // The AllowsUnaligned flag models the SCTLR.A setting in ARM cpus
19154 bool AllowsUnaligned = Subtarget->allowsUnalignedMem();
19155 auto Ty = VT.getSimpleVT().SimpleTy;
19156
19157 if (Ty == MVT::i8 || Ty == MVT::i16 || Ty == MVT::i32) {
19158 // Unaligned access can use (for example) LRDB, LRDH, LDR
19159 if (AllowsUnaligned) {
19160 if (Fast)
19161 *Fast = Subtarget->hasV7Ops();
19162 return true;
19163 }
19164 }
19165
19166 if (Ty == MVT::f64 || Ty == MVT::v2f64) {
19167 // For any little-endian targets with neon, we can support unaligned ld/st
19168 // of D and Q (e.g. {D0,D1}) registers by using vld1.i8/vst1.i8.
19169 // A big-endian target may also explicitly support unaligned accesses
19170 if (Subtarget->hasNEON() && (AllowsUnaligned || Subtarget->isLittle())) {
19171 if (Fast)
19172 *Fast = 1;
19173 return true;
19174 }
19175 }
19176
19177 if (!Subtarget->hasMVEIntegerOps())
19178 return false;
19179
19180 // These are for predicates
19181 if ((Ty == MVT::v16i1 || Ty == MVT::v8i1 || Ty == MVT::v4i1 ||
19182 Ty == MVT::v2i1)) {
19183 if (Fast)
19184 *Fast = 1;
19185 return true;
19186 }
19187
19188 // These are for truncated stores/narrowing loads. They are fine so long as
19189 // the alignment is at least the size of the item being loaded
19190 if ((Ty == MVT::v4i8 || Ty == MVT::v8i8 || Ty == MVT::v4i16) &&
19191 Alignment >= VT.getScalarSizeInBits() / 8) {
19192 if (Fast)
19193 *Fast = true;
19194 return true;
19195 }
19196
19197 // In little-endian MVE, the store instructions VSTRB.U8, VSTRH.U16 and
19198 // VSTRW.U32 all store the vector register in exactly the same format, and
19199 // differ only in the range of their immediate offset field and the required
19200 // alignment. So there is always a store that can be used, regardless of
19201 // actual type.
19202 //
19203 // For big endian, that is not the case. But can still emit a (VSTRB.U8;
19204 // VREV64.8) pair and get the same effect. This will likely be better than
19205 // aligning the vector through the stack.
19206 if (Ty == MVT::v16i8 || Ty == MVT::v8i16 || Ty == MVT::v8f16 ||
19207 Ty == MVT::v4i32 || Ty == MVT::v4f32 || Ty == MVT::v2i64 ||
19208 Ty == MVT::v2f64) {
19209 if (Fast)
19210 *Fast = 1;
19211 return true;
19212 }
19213
19214 return false;
19215}
19216
19218 LLVMContext &Context, const MemOp &Op,
19219 const AttributeList &FuncAttributes) const {
19220 // See if we can use NEON instructions for this...
19221 if ((Op.isMemcpyOrMemmove() || Op.isZeroMemset()) && Subtarget->hasNEON() &&
19222 !FuncAttributes.hasFnAttr(Attribute::NoImplicitFloat)) {
19223 unsigned Fast;
19224 if (Op.size() >= 16 &&
19225 (Op.isAligned(Align(16)) ||
19226 (allowsMisalignedMemoryAccesses(MVT::v2f64, 0, Align(1),
19228 Fast))) {
19229 return MVT::v2f64;
19230 } else if (Op.size() >= 8 &&
19231 (Op.isAligned(Align(8)) ||
19233 MVT::f64, 0, Align(1), MachineMemOperand::MONone, &Fast) &&
19234 Fast))) {
19235 return MVT::f64;
19236 }
19237 }
19238
19239 // Let the target-independent logic figure it out.
19240 return MVT::Other;
19241}
19242
19243// 64-bit integers are split into their high and low parts and held in two
19244// different registers, so the trunc is free since the low register can just
19245// be used.
19246bool ARMTargetLowering::isTruncateFree(Type *SrcTy, Type *DstTy) const {
19247 if (!SrcTy->isIntegerTy() || !DstTy->isIntegerTy())
19248 return false;
19249 unsigned SrcBits = SrcTy->getPrimitiveSizeInBits();
19250 unsigned DestBits = DstTy->getPrimitiveSizeInBits();
19251 return (SrcBits == 64 && DestBits == 32);
19252}
19253
19255 if (SrcVT.isVector() || DstVT.isVector() || !SrcVT.isInteger() ||
19256 !DstVT.isInteger())
19257 return false;
19258 unsigned SrcBits = SrcVT.getSizeInBits();
19259 unsigned DestBits = DstVT.getSizeInBits();
19260 return (SrcBits == 64 && DestBits == 32);
19261}
19262
19264 if (Val.getOpcode() != ISD::LOAD)
19265 return false;
19266
19267 EVT VT1 = Val.getValueType();
19268 if (!VT1.isSimple() || !VT1.isInteger() ||
19269 !VT2.isSimple() || !VT2.isInteger())
19270 return false;
19271
19272 switch (VT1.getSimpleVT().SimpleTy) {
19273 default: break;
19274 case MVT::i1:
19275 case MVT::i8:
19276 case MVT::i16:
19277 // 8-bit and 16-bit loads implicitly zero-extend to 32-bits.
19278 return true;
19279 }
19280
19281 return false;
19282}
19283
19285 if (!VT.isSimple())
19286 return false;
19287
19288 // There are quite a few FP16 instructions (e.g. VNMLA, VNMLS, etc.) that
19289 // negate values directly (fneg is free). So, we don't want to let the DAG
19290 // combiner rewrite fneg into xors and some other instructions. For f16 and
19291 // FullFP16 argument passing, some bitcast nodes may be introduced,
19292 // triggering this DAG combine rewrite, so we are avoiding that with this.
19293 switch (VT.getSimpleVT().SimpleTy) {
19294 default: break;
19295 case MVT::f16:
19296 return Subtarget->hasFullFP16();
19297 }
19298
19299 return false;
19300}
19301
19303 if (!Subtarget->hasMVEIntegerOps())
19304 return nullptr;
19305 Type *SVIType = SVI->getType();
19306 Type *ScalarType = SVIType->getScalarType();
19307
19308 if (ScalarType->isFloatTy())
19309 return Type::getInt32Ty(SVIType->getContext());
19310 if (ScalarType->isHalfTy())
19311 return Type::getInt16Ty(SVIType->getContext());
19312 return nullptr;
19313}
19314
19316 EVT VT = ExtVal.getValueType();
19317
19318 if (!isTypeLegal(VT))
19319 return false;
19320
19321 if (auto *Ld = dyn_cast<MaskedLoadSDNode>(ExtVal.getOperand(0))) {
19322 if (Ld->isExpandingLoad())
19323 return false;
19324 }
19325
19326 if (Subtarget->hasMVEIntegerOps())
19327 return true;
19328
19329 // Don't create a loadext if we can fold the extension into a wide/long
19330 // instruction.
19331 // If there's more than one user instruction, the loadext is desirable no
19332 // matter what. There can be two uses by the same instruction.
19333 if (ExtVal->use_empty() ||
19334 !ExtVal->user_begin()->isOnlyUserOf(ExtVal.getNode()))
19335 return true;
19336
19337 SDNode *U = *ExtVal->user_begin();
19338 if ((U->getOpcode() == ISD::ADD || U->getOpcode() == ISD::SUB ||
19339 U->getOpcode() == ISD::SHL || U->getOpcode() == ARMISD::VSHLIMM))
19340 return false;
19341
19342 return true;
19343}
19344
19346 if (!Ty1->isIntegerTy() || !Ty2->isIntegerTy())
19347 return false;
19348
19349 if (!isTypeLegal(EVT::getEVT(Ty1)))
19350 return false;
19351
19352 assert(Ty1->getPrimitiveSizeInBits() <= 64 && "i128 is probably not a noop");
19353
19354 // Assuming the caller doesn't have a zeroext or signext return parameter,
19355 // truncation all the way down to i1 is valid.
19356 return true;
19357}
19358
19359/// isFMAFasterThanFMulAndFAdd - Return true if an FMA operation is faster
19360/// than a pair of fmul and fadd instructions. fmuladd intrinsics will be
19361/// expanded to FMAs when this method returns true, otherwise fmuladd is
19362/// expanded to fmul + fadd.
19363///
19364/// ARM supports both fused and unfused multiply-add operations; we already
19365/// lower a pair of fmul and fadd to the latter so it's not clear that there
19366/// would be a gain or that the gain would be worthwhile enough to risk
19367/// correctness bugs.
19368///
19369/// For MVE, we set this to true as it helps simplify the need for some
19370/// patterns (and we don't have the non-fused floating point instruction).
19371bool ARMTargetLowering::isFMAFasterThanFMulAndFAdd(const MachineFunction &MF,
19372 EVT VT) const {
19373 if (Subtarget->useSoftFloat())
19374 return false;
19375
19376 if (!VT.isSimple())
19377 return false;
19378
19379 switch (VT.getSimpleVT().SimpleTy) {
19380 case MVT::v4f32:
19381 case MVT::v8f16:
19382 return Subtarget->hasMVEFloatOps();
19383 case MVT::f16:
19384 return Subtarget->useFPVFMx16();
19385 case MVT::f32:
19386 return Subtarget->useFPVFMx();
19387 case MVT::f64:
19388 return Subtarget->useFPVFMx64();
19389 default:
19390 break;
19391 }
19392
19393 return false;
19394}
19395
19396static bool isLegalT1AddressImmediate(int64_t V, EVT VT) {
19397 if (V < 0)
19398 return false;
19399
19400 unsigned Scale = 1;
19401 switch (VT.getSimpleVT().SimpleTy) {
19402 case MVT::i1:
19403 case MVT::i8:
19404 // Scale == 1;
19405 break;
19406 case MVT::i16:
19407 // Scale == 2;
19408 Scale = 2;
19409 break;
19410 default:
19411 // On thumb1 we load most things (i32, i64, floats, etc) with a LDR
19412 // Scale == 4;
19413 Scale = 4;
19414 break;
19415 }
19416
19417 if ((V & (Scale - 1)) != 0)
19418 return false;
19419 return isUInt<5>(V / Scale);
19420}
19421
19422static bool isLegalT2AddressImmediate(int64_t V, EVT VT,
19423 const ARMSubtarget *Subtarget) {
19424 if (!VT.isInteger() && !VT.isFloatingPoint())
19425 return false;
19426 if (VT.isVector() && Subtarget->hasNEON())
19427 return false;
19428 if (VT.isVector() && VT.isFloatingPoint() && Subtarget->hasMVEIntegerOps() &&
19429 !Subtarget->hasMVEFloatOps())
19430 return false;
19431
19432 bool IsNeg = false;
19433 if (V < 0) {
19434 IsNeg = true;
19435 V = -V;
19436 }
19437
19438 unsigned NumBytes = std::max((unsigned)VT.getSizeInBits() / 8, 1U);
19439
19440 // MVE: size * imm7
19441 if (VT.isVector() && Subtarget->hasMVEIntegerOps()) {
19442 switch (VT.getSimpleVT().getVectorElementType().SimpleTy) {
19443 case MVT::i32:
19444 case MVT::f32:
19445 return isShiftedUInt<7,2>(V);
19446 case MVT::i16:
19447 case MVT::f16:
19448 return isShiftedUInt<7,1>(V);
19449 case MVT::i8:
19450 return isUInt<7>(V);
19451 default:
19452 return false;
19453 }
19454 }
19455
19456 // half VLDR: 2 * imm8
19457 if (VT.isFloatingPoint() && NumBytes == 2 && Subtarget->hasFPRegs16())
19458 return isShiftedUInt<8, 1>(V);
19459 // VLDR and LDRD: 4 * imm8
19460 if ((VT.isFloatingPoint() && Subtarget->hasVFP2Base()) || NumBytes == 8)
19461 return isShiftedUInt<8, 2>(V);
19462
19463 if (NumBytes == 1 || NumBytes == 2 || NumBytes == 4) {
19464 // + imm12 or - imm8
19465 if (IsNeg)
19466 return isUInt<8>(V);
19467 return isUInt<12>(V);
19468 }
19469
19470 return false;
19471}
19472
19473/// isLegalAddressImmediate - Return true if the integer value can be used
19474/// as the offset of the target addressing mode for load / store of the
19475/// given type.
19476static bool isLegalAddressImmediate(int64_t V, EVT VT,
19477 const ARMSubtarget *Subtarget) {
19478 if (V == 0)
19479 return true;
19480
19481 if (!VT.isSimple())
19482 return false;
19483
19484 if (Subtarget->isThumb1Only())
19485 return isLegalT1AddressImmediate(V, VT);
19486 else if (Subtarget->isThumb2())
19487 return isLegalT2AddressImmediate(V, VT, Subtarget);
19488
19489 // ARM mode.
19490 if (V < 0)
19491 V = - V;
19492 switch (VT.getSimpleVT().SimpleTy) {
19493 default: return false;
19494 case MVT::i1:
19495 case MVT::i8:
19496 case MVT::i32:
19497 // +- imm12
19498 return isUInt<12>(V);
19499 case MVT::i16:
19500 // +- imm8
19501 return isUInt<8>(V);
19502 case MVT::f32:
19503 case MVT::f64:
19504 if (!Subtarget->hasVFP2Base()) // FIXME: NEON?
19505 return false;
19506 return isShiftedUInt<8, 2>(V);
19507 }
19508}
19509
19511 EVT VT) const {
19512 int Scale = AM.Scale;
19513 if (Scale < 0)
19514 return false;
19515
19516 switch (VT.getSimpleVT().SimpleTy) {
19517 default: return false;
19518 case MVT::i1:
19519 case MVT::i8:
19520 case MVT::i16:
19521 case MVT::i32:
19522 if (Scale == 1)
19523 return true;
19524 // r + r << imm
19525 Scale = Scale & ~1;
19526 return Scale == 2 || Scale == 4 || Scale == 8;
19527 case MVT::i64:
19528 // FIXME: What are we trying to model here? ldrd doesn't have an r + r
19529 // version in Thumb mode.
19530 // r + r
19531 if (Scale == 1)
19532 return true;
19533 // r * 2 (this can be lowered to r + r).
19534 if (!AM.HasBaseReg && Scale == 2)
19535 return true;
19536 return false;
19537 case MVT::isVoid:
19538 // Note, we allow "void" uses (basically, uses that aren't loads or
19539 // stores), because arm allows folding a scale into many arithmetic
19540 // operations. This should be made more precise and revisited later.
19541
19542 // Allow r << imm, but the imm has to be a multiple of two.
19543 if (Scale & 1) return false;
19544 return isPowerOf2_32(Scale);
19545 }
19546}
19547
19549 EVT VT) const {
19550 const int Scale = AM.Scale;
19551
19552 // Negative scales are not supported in Thumb1.
19553 if (Scale < 0)
19554 return false;
19555
19556 // Thumb1 addressing modes do not support register scaling excepting the
19557 // following cases:
19558 // 1. Scale == 1 means no scaling.
19559 // 2. Scale == 2 this can be lowered to r + r if there is no base register.
19560 return (Scale == 1) || (!AM.HasBaseReg && Scale == 2);
19561}
19562
19563/// isLegalAddressingMode - Return true if the addressing mode represented
19564/// by AM is legal for this target, for a load/store of the specified type.
19566 const AddrMode &AM, Type *Ty,
19567 unsigned AS, Instruction *I) const {
19568 EVT VT = getValueType(DL, Ty, true);
19569 if (!isLegalAddressImmediate(AM.BaseOffs, VT, Subtarget))
19570 return false;
19571
19572 // Can never fold addr of global into load/store.
19573 if (AM.BaseGV)
19574 return false;
19575
19576 switch (AM.Scale) {
19577 case 0: // no scale reg, must be "r+i" or "r", or "i".
19578 break;
19579 default:
19580 // ARM doesn't support any R+R*scale+imm addr modes.
19581 if (AM.BaseOffs)
19582 return false;
19583
19584 if (!VT.isSimple())
19585 return false;
19586
19587 if (Subtarget->isThumb1Only())
19588 return isLegalT1ScaledAddressingMode(AM, VT);
19589
19590 if (Subtarget->isThumb2())
19591 return isLegalT2ScaledAddressingMode(AM, VT);
19592
19593 int Scale = AM.Scale;
19594 switch (VT.getSimpleVT().SimpleTy) {
19595 default: return false;
19596 case MVT::i1:
19597 case MVT::i8:
19598 case MVT::i32:
19599 if (Scale < 0) Scale = -Scale;
19600 if (Scale == 1)
19601 return true;
19602 // r + r << imm
19603 return isPowerOf2_32(Scale & ~1);
19604 case MVT::i16:
19605 case MVT::i64:
19606 // r +/- r
19607 if (Scale == 1 || (AM.HasBaseReg && Scale == -1))
19608 return true;
19609 // r * 2 (this can be lowered to r + r).
19610 if (!AM.HasBaseReg && Scale == 2)
19611 return true;
19612 return false;
19613
19614 case MVT::isVoid:
19615 // Note, we allow "void" uses (basically, uses that aren't loads or
19616 // stores), because arm allows folding a scale into many arithmetic
19617 // operations. This should be made more precise and revisited later.
19618
19619 // Allow r << imm, but the imm has to be a multiple of two.
19620 if (Scale & 1) return false;
19621 return isPowerOf2_32(Scale);
19622 }
19623 }
19624 return true;
19625}
19626
19627/// isLegalICmpImmediate - Return true if the specified immediate is legal
19628/// icmp immediate, that is the target has icmp instructions which can compare
19629/// a register against the immediate without having to materialize the
19630/// immediate into a register.
19632 // Thumb2 and ARM modes can use cmn for negative immediates.
19633 if (!Subtarget->isThumb())
19634 return ARM_AM::getSOImmVal((uint32_t)Imm) != -1 ||
19636 if (Subtarget->isThumb2())
19637 return ARM_AM::getT2SOImmVal((uint32_t)Imm) != -1 ||
19639 // Thumb1 doesn't have cmn, and only 8-bit immediates.
19640 return Imm >= 0 && Imm <= 255;
19641}
19642
19643/// isLegalAddImmediate - Return true if the specified immediate is a legal add
19644/// *or sub* immediate, that is the target has add or sub instructions which can
19645/// add a register with the immediate without having to materialize the
19646/// immediate into a register.
19648 // Same encoding for add/sub, just flip the sign.
19649 uint64_t AbsImm = AbsoluteValue(Imm);
19650 if (!Subtarget->isThumb())
19651 return ARM_AM::getSOImmVal(AbsImm) != -1;
19652 if (Subtarget->isThumb2())
19653 return ARM_AM::getT2SOImmVal(AbsImm) != -1;
19654 // Thumb1 only has 8-bit unsigned immediate.
19655 return AbsImm <= 255;
19656}
19657
19658// Return false to prevent folding
19659// (mul (add r, c0), c1) -> (add (mul r, c1), c0*c1) in DAGCombine,
19660// if the folding leads to worse code.
19662 SDValue ConstNode) const {
19663 // Let the DAGCombiner decide for vector types and large types.
19664 const EVT VT = AddNode.getValueType();
19665 if (VT.isVector() || VT.getScalarSizeInBits() > 32)
19666 return true;
19667
19668 // It is worse if c0 is legal add immediate, while c1*c0 is not
19669 // and has to be composed by at least two instructions.
19670 const ConstantSDNode *C0Node = cast<ConstantSDNode>(AddNode.getOperand(1));
19671 const ConstantSDNode *C1Node = cast<ConstantSDNode>(ConstNode);
19672 const int64_t C0 = C0Node->getSExtValue();
19673 APInt CA = C0Node->getAPIntValue() * C1Node->getAPIntValue();
19675 return true;
19676 if (ConstantMaterializationCost((unsigned)CA.getZExtValue(), Subtarget) > 1)
19677 return false;
19678
19679 // Default to true and let the DAGCombiner decide.
19680 return true;
19681}
19682
19684 bool isSEXTLoad, SDValue &Base,
19685 SDValue &Offset, bool &isInc,
19686 SelectionDAG &DAG) {
19687 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB)
19688 return false;
19689
19690 if (VT == MVT::i16 || ((VT == MVT::i8 || VT == MVT::i1) && isSEXTLoad)) {
19691 // AddressingMode 3
19692 Base = Ptr->getOperand(0);
19694 int RHSC = (int)RHS->getZExtValue();
19695 if (RHSC < 0 && RHSC > -256) {
19696 assert(Ptr->getOpcode() == ISD::ADD);
19697 isInc = false;
19698 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0));
19699 return true;
19700 }
19701 }
19702 isInc = (Ptr->getOpcode() == ISD::ADD);
19703 Offset = Ptr->getOperand(1);
19704 return true;
19705 } else if (VT == MVT::i32 || VT == MVT::i8 || VT == MVT::i1) {
19706 // AddressingMode 2
19708 int RHSC = (int)RHS->getZExtValue();
19709 if (RHSC < 0 && RHSC > -0x1000) {
19710 assert(Ptr->getOpcode() == ISD::ADD);
19711 isInc = false;
19712 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0));
19713 Base = Ptr->getOperand(0);
19714 return true;
19715 }
19716 }
19717
19718 if (Ptr->getOpcode() == ISD::ADD) {
19719 isInc = true;
19720 ARM_AM::ShiftOpc ShOpcVal=
19722 if (ShOpcVal != ARM_AM::no_shift) {
19723 Base = Ptr->getOperand(1);
19724 Offset = Ptr->getOperand(0);
19725 } else {
19726 Base = Ptr->getOperand(0);
19727 Offset = Ptr->getOperand(1);
19728 }
19729 return true;
19730 }
19731
19732 isInc = (Ptr->getOpcode() == ISD::ADD);
19733 Base = Ptr->getOperand(0);
19734 Offset = Ptr->getOperand(1);
19735 return true;
19736 }
19737
19738 // FIXME: Use VLDM / VSTM to emulate indexed FP load / store.
19739 return false;
19740}
19741
19743 bool isSEXTLoad, SDValue &Base,
19744 SDValue &Offset, bool &isInc,
19745 SelectionDAG &DAG) {
19746 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB)
19747 return false;
19748
19749 Base = Ptr->getOperand(0);
19751 int RHSC = (int)RHS->getZExtValue();
19752 if (RHSC < 0 && RHSC > -0x100) { // 8 bits.
19753 assert(Ptr->getOpcode() == ISD::ADD);
19754 isInc = false;
19755 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0));
19756 return true;
19757 } else if (RHSC > 0 && RHSC < 0x100) { // 8 bit, no zero.
19758 isInc = Ptr->getOpcode() == ISD::ADD;
19759 Offset = DAG.getConstant(RHSC, SDLoc(Ptr), RHS->getValueType(0));
19760 return true;
19761 }
19762 }
19763
19764 return false;
19765}
19766
19767static bool getMVEIndexedAddressParts(SDNode *Ptr, EVT VT, Align Alignment,
19768 bool isSEXTLoad, bool IsMasked, bool isLE,
19770 bool &isInc, SelectionDAG &DAG) {
19771 if (Ptr->getOpcode() != ISD::ADD && Ptr->getOpcode() != ISD::SUB)
19772 return false;
19773 if (!isa<ConstantSDNode>(Ptr->getOperand(1)))
19774 return false;
19775
19776 // We allow LE non-masked loads to change the type (for example use a vldrb.8
19777 // as opposed to a vldrw.32). This can allow extra addressing modes or
19778 // alignments for what is otherwise an equivalent instruction.
19779 bool CanChangeType = isLE && !IsMasked;
19780
19782 int RHSC = (int)RHS->getZExtValue();
19783
19784 auto IsInRange = [&](int RHSC, int Limit, int Scale) {
19785 if (RHSC < 0 && RHSC > -Limit * Scale && RHSC % Scale == 0) {
19786 assert(Ptr->getOpcode() == ISD::ADD);
19787 isInc = false;
19788 Offset = DAG.getConstant(-RHSC, SDLoc(Ptr), RHS->getValueType(0));
19789 return true;
19790 } else if (RHSC > 0 && RHSC < Limit * Scale && RHSC % Scale == 0) {
19791 isInc = Ptr->getOpcode() == ISD::ADD;
19792 Offset = DAG.getConstant(RHSC, SDLoc(Ptr), RHS->getValueType(0));
19793 return true;
19794 }
19795 return false;
19796 };
19797
19798 // Try to find a matching instruction based on s/zext, Alignment, Offset and
19799 // (in BE/masked) type.
19800 Base = Ptr->getOperand(0);
19801 if (VT == MVT::v4i16) {
19802 if (Alignment >= 2 && IsInRange(RHSC, 0x80, 2))
19803 return true;
19804 } else if (VT == MVT::v4i8 || VT == MVT::v8i8) {
19805 if (IsInRange(RHSC, 0x80, 1))
19806 return true;
19807 } else if (Alignment >= 4 &&
19808 (CanChangeType || VT == MVT::v4i32 || VT == MVT::v4f32) &&
19809 IsInRange(RHSC, 0x80, 4))
19810 return true;
19811 else if (Alignment >= 2 &&
19812 (CanChangeType || VT == MVT::v8i16 || VT == MVT::v8f16) &&
19813 IsInRange(RHSC, 0x80, 2))
19814 return true;
19815 else if ((CanChangeType || VT == MVT::v16i8) && IsInRange(RHSC, 0x80, 1))
19816 return true;
19817 return false;
19818}
19819
19820/// getPreIndexedAddressParts - returns true by value, base pointer and
19821/// offset pointer and addressing mode by reference if the node's address
19822/// can be legally represented as pre-indexed load / store address.
19823bool
19825 SDValue &Offset,
19827 SelectionDAG &DAG) const {
19828 if (Subtarget->isThumb1Only())
19829 return false;
19830
19831 EVT VT;
19832 SDValue Ptr;
19833 Align Alignment;
19834 unsigned AS = 0;
19835 bool isSEXTLoad = false;
19836 bool IsMasked = false;
19837 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) {
19838 Ptr = LD->getBasePtr();
19839 VT = LD->getMemoryVT();
19840 Alignment = LD->getAlign();
19841 AS = LD->getAddressSpace();
19842 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD;
19843 } else if (StoreSDNode *ST = dyn_cast<StoreSDNode>(N)) {
19844 Ptr = ST->getBasePtr();
19845 VT = ST->getMemoryVT();
19846 Alignment = ST->getAlign();
19847 AS = ST->getAddressSpace();
19848 } else if (MaskedLoadSDNode *LD = dyn_cast<MaskedLoadSDNode>(N)) {
19849 Ptr = LD->getBasePtr();
19850 VT = LD->getMemoryVT();
19851 Alignment = LD->getAlign();
19852 AS = LD->getAddressSpace();
19853 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD;
19854 IsMasked = true;
19856 Ptr = ST->getBasePtr();
19857 VT = ST->getMemoryVT();
19858 Alignment = ST->getAlign();
19859 AS = ST->getAddressSpace();
19860 IsMasked = true;
19861 } else
19862 return false;
19863
19864 unsigned Fast = 0;
19865 if (!allowsMisalignedMemoryAccesses(VT, AS, Alignment,
19867 // Only generate post-increment or pre-increment forms when a real
19868 // hardware instruction exists for them. Do not emit postinc/preinc
19869 // if the operation will end up as a libcall.
19870 return false;
19871 }
19872
19873 bool isInc;
19874 bool isLegal = false;
19875 if (VT.isVector())
19876 isLegal = Subtarget->hasMVEIntegerOps() &&
19878 Ptr.getNode(), VT, Alignment, isSEXTLoad, IsMasked,
19879 Subtarget->isLittle(), Base, Offset, isInc, DAG);
19880 else {
19881 if (Subtarget->isThumb2())
19882 isLegal = getT2IndexedAddressParts(Ptr.getNode(), VT, isSEXTLoad, Base,
19883 Offset, isInc, DAG);
19884 else
19885 isLegal = getARMIndexedAddressParts(Ptr.getNode(), VT, isSEXTLoad, Base,
19886 Offset, isInc, DAG);
19887 }
19888 if (!isLegal)
19889 return false;
19890
19891 AM = isInc ? ISD::PRE_INC : ISD::PRE_DEC;
19892 return true;
19893}
19894
19895/// getPostIndexedAddressParts - returns true by value, base pointer and
19896/// offset pointer and addressing mode by reference if this node can be
19897/// combined with a load / store to form a post-indexed load / store.
19899 SDValue &Base,
19900 SDValue &Offset,
19902 SelectionDAG &DAG) const {
19903 EVT VT;
19904 SDValue Ptr;
19905 Align Alignment;
19906 bool isSEXTLoad = false, isNonExt;
19907 bool IsMasked = false;
19908 if (LoadSDNode *LD = dyn_cast<LoadSDNode>(N)) {
19909 VT = LD->getMemoryVT();
19910 Ptr = LD->getBasePtr();
19911 Alignment = LD->getAlign();
19912 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD;
19913 isNonExt = LD->getExtensionType() == ISD::NON_EXTLOAD;
19914 } else if (StoreSDNode *ST = dyn_cast<StoreSDNode>(N)) {
19915 VT = ST->getMemoryVT();
19916 Ptr = ST->getBasePtr();
19917 Alignment = ST->getAlign();
19918 isNonExt = !ST->isTruncatingStore();
19919 } else if (MaskedLoadSDNode *LD = dyn_cast<MaskedLoadSDNode>(N)) {
19920 VT = LD->getMemoryVT();
19921 Ptr = LD->getBasePtr();
19922 Alignment = LD->getAlign();
19923 isSEXTLoad = LD->getExtensionType() == ISD::SEXTLOAD;
19924 isNonExt = LD->getExtensionType() == ISD::NON_EXTLOAD;
19925 IsMasked = true;
19927 VT = ST->getMemoryVT();
19928 Ptr = ST->getBasePtr();
19929 Alignment = ST->getAlign();
19930 isNonExt = !ST->isTruncatingStore();
19931 IsMasked = true;
19932 } else
19933 return false;
19934
19935 if (Subtarget->isThumb1Only()) {
19936 // Thumb-1 can do a limited post-inc load or store as an updating LDM. It
19937 // must be non-extending/truncating, i32, with an offset of 4.
19938 assert(Op->getValueType(0) == MVT::i32 && "Non-i32 post-inc op?!");
19939 if (Op->getOpcode() != ISD::ADD || !isNonExt)
19940 return false;
19941 auto *RHS = dyn_cast<ConstantSDNode>(Op->getOperand(1));
19942 if (!RHS || RHS->getZExtValue() != 4)
19943 return false;
19944 if (Alignment < Align(4))
19945 return false;
19946
19947 Offset = Op->getOperand(1);
19948 Base = Op->getOperand(0);
19949 AM = ISD::POST_INC;
19950 return true;
19951 }
19952
19953 bool isInc;
19954 bool isLegal = false;
19955 if (VT.isVector())
19956 isLegal = Subtarget->hasMVEIntegerOps() &&
19957 getMVEIndexedAddressParts(Op, VT, Alignment, isSEXTLoad, IsMasked,
19958 Subtarget->isLittle(), Base, Offset,
19959 isInc, DAG);
19960 else {
19961 if (Subtarget->isThumb2())
19962 isLegal = getT2IndexedAddressParts(Op, VT, isSEXTLoad, Base, Offset,
19963 isInc, DAG);
19964 else
19965 isLegal = getARMIndexedAddressParts(Op, VT, isSEXTLoad, Base, Offset,
19966 isInc, DAG);
19967 }
19968 if (!isLegal)
19969 return false;
19970
19971 if (Ptr != Base) {
19972 // Swap base ptr and offset to catch more post-index load / store when
19973 // it's legal. In Thumb2 mode, offset must be an immediate.
19974 if (Ptr == Offset && Op->getOpcode() == ISD::ADD &&
19975 !Subtarget->isThumb2())
19977
19978 // Post-indexed load / store update the base pointer.
19979 if (Ptr != Base)
19980 return false;
19981 }
19982
19983 AM = isInc ? ISD::POST_INC : ISD::POST_DEC;
19984 return true;
19985}
19986
19989 const APInt &DemandedElts,
19990 const SelectionDAG &DAG,
19991 unsigned Depth) const {
19992 unsigned BitWidth = Known.getBitWidth();
19993 Known.resetAll();
19994 switch (Op.getOpcode()) {
19995 default: break;
19996 case ARMISD::ADDC:
19997 case ARMISD::ADDE:
19998 case ARMISD::SUBC:
19999 case ARMISD::SUBE:
20000 // Special cases when we convert a carry to a boolean.
20001 if (Op.getResNo() == 0) {
20002 SDValue LHS = Op.getOperand(0);
20003 SDValue RHS = Op.getOperand(1);
20004 // (ADDE 0, 0, C) will give us a single bit.
20005 if (Op->getOpcode() == ARMISD::ADDE && isNullConstant(LHS) &&
20006 isNullConstant(RHS)) {
20008 return;
20009 }
20010 }
20011 break;
20012 case ARMISD::CMOV: {
20013 // Bits are known zero/one if known on the LHS and RHS.
20014 Known = DAG.computeKnownBits(Op.getOperand(0), Depth+1);
20015 if (Known.isUnknown())
20016 return;
20017
20018 KnownBits KnownRHS = DAG.computeKnownBits(Op.getOperand(1), Depth+1);
20019 Known = Known.intersectWith(KnownRHS);
20020 return;
20021 }
20023 Intrinsic::ID IntID =
20024 static_cast<Intrinsic::ID>(Op->getConstantOperandVal(1));
20025 switch (IntID) {
20026 default: return;
20027 case Intrinsic::arm_ldaex:
20028 case Intrinsic::arm_ldrex: {
20029 EVT VT = cast<MemIntrinsicSDNode>(Op)->getMemoryVT();
20030 unsigned MemBits = VT.getScalarSizeInBits();
20031 Known.Zero |= APInt::getHighBitsSet(BitWidth, BitWidth - MemBits);
20032 return;
20033 }
20034 }
20035 }
20036 case ARMISD::BFI: {
20037 // Conservatively, we can recurse down the first operand
20038 // and just mask out all affected bits.
20039 Known = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
20040
20041 // The operand to BFI is already a mask suitable for removing the bits it
20042 // sets.
20043 const APInt &Mask = Op.getConstantOperandAPInt(2);
20044 Known.Zero &= Mask;
20045 Known.One &= Mask;
20046 return;
20047 }
20048 case ARMISD::VGETLANEs:
20049 case ARMISD::VGETLANEu: {
20050 const SDValue &SrcSV = Op.getOperand(0);
20051 EVT VecVT = SrcSV.getValueType();
20052 assert(VecVT.isVector() && "VGETLANE expected a vector type");
20053 const unsigned NumSrcElts = VecVT.getVectorNumElements();
20054 ConstantSDNode *Pos = cast<ConstantSDNode>(Op.getOperand(1).getNode());
20055 assert(Pos->getAPIntValue().ult(NumSrcElts) &&
20056 "VGETLANE index out of bounds");
20057 unsigned Idx = Pos->getZExtValue();
20058 APInt DemandedElt = APInt::getOneBitSet(NumSrcElts, Idx);
20059 Known = DAG.computeKnownBits(SrcSV, DemandedElt, Depth + 1);
20060
20061 EVT VT = Op.getValueType();
20062 const unsigned DstSz = VT.getScalarSizeInBits();
20063 const unsigned SrcSz = VecVT.getVectorElementType().getSizeInBits();
20064 (void)SrcSz;
20065 assert(SrcSz == Known.getBitWidth());
20066 assert(DstSz > SrcSz);
20067 if (Op.getOpcode() == ARMISD::VGETLANEs)
20068 Known = Known.sext(DstSz);
20069 else {
20070 Known = Known.zext(DstSz);
20071 }
20072 assert(DstSz == Known.getBitWidth());
20073 break;
20074 }
20075 case ARMISD::VMOVrh: {
20076 KnownBits KnownOp = DAG.computeKnownBits(Op->getOperand(0), Depth + 1);
20077 assert(KnownOp.getBitWidth() == 16);
20078 Known = KnownOp.zext(32);
20079 break;
20080 }
20081 case ARMISD::CSINC:
20082 case ARMISD::CSINV:
20083 case ARMISD::CSNEG: {
20084 KnownBits KnownOp0 = DAG.computeKnownBits(Op->getOperand(0), Depth + 1);
20085 KnownBits KnownOp1 = DAG.computeKnownBits(Op->getOperand(1), Depth + 1);
20086
20087 // The result is either:
20088 // CSINC: KnownOp0 or KnownOp1 + 1
20089 // CSINV: KnownOp0 or ~KnownOp1
20090 // CSNEG: KnownOp0 or KnownOp1 * -1
20091 if (Op.getOpcode() == ARMISD::CSINC)
20092 KnownOp1 =
20093 KnownBits::add(KnownOp1, KnownBits::makeConstant(APInt(32, 1)));
20094 else if (Op.getOpcode() == ARMISD::CSINV)
20095 std::swap(KnownOp1.Zero, KnownOp1.One);
20096 else if (Op.getOpcode() == ARMISD::CSNEG)
20097 KnownOp1 = KnownBits::mul(KnownOp1,
20099
20100 Known = KnownOp0.intersectWith(KnownOp1);
20101 break;
20102 }
20103 case ARMISD::VORRIMM:
20104 case ARMISD::VBICIMM: {
20105 unsigned Encoded = Op.getConstantOperandVal(1);
20106 unsigned DecEltBits = 0;
20107 uint64_t DecodedVal = ARM_AM::decodeVMOVModImm(Encoded, DecEltBits);
20108
20109 unsigned EltBits = Op.getScalarValueSizeInBits();
20110 if (EltBits != DecEltBits) {
20111 // Be conservative: only update Known when EltBits == DecEltBits.
20112 // This is believed to always be true for VORRIMM/VBICIMM today, but if
20113 // that changes in the future, doing nothing here is safer than risking
20114 // subtle bugs.
20115 break;
20116 }
20117
20118 KnownBits KnownLHS = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
20119 bool IsVORR = Op.getOpcode() == ARMISD::VORRIMM;
20120 APInt Imm(DecEltBits, DecodedVal);
20121
20122 Known.One = IsVORR ? (KnownLHS.One | Imm) : (KnownLHS.One & ~Imm);
20123 Known.Zero = IsVORR ? (KnownLHS.Zero & ~Imm) : (KnownLHS.Zero | Imm);
20124 break;
20125 }
20126 }
20127}
20128
20129static bool isLegalLogicalImmediate(unsigned Imm,
20130 const ARMSubtarget *Subtarget) {
20131 if (!Subtarget->isThumb())
20132 return ARM_AM::getSOImmVal(Imm) != -1;
20133 if (Subtarget->isThumb2())
20134 return ARM_AM::getT2SOImmVal(Imm) != -1;
20135 // Thumb1 only has 8-bit unsigned immediate.
20136 return Imm <= 255;
20137}
20138
20139/// Refine i32 AND/OR/XOR with a constant RHS using demanded bits: replace the
20140/// immediate with an equivalent constant that ARM/Thumb can encode as a
20141/// logical immediate (or that selects better lowering), without changing the
20142/// computed result on those demanded bits.
20143static bool optimizeLogicalImm(SDValue Op, unsigned Imm,
20144 const APInt &DemandedBits,
20145 const ARMSubtarget *Subtarget,
20147
20148 if (Imm == 0 || Imm == ~0U)
20149 return false;
20150
20151 unsigned Opc = Op.getOpcode();
20152 unsigned Demanded = DemandedBits.getZExtValue();
20153 EVT VT = Op.getValueType();
20154
20155 unsigned ShrunkImm = Imm & Demanded;
20156 unsigned ExpandedImm = Imm | ~Demanded;
20157
20158 auto IsLegalImm = [ShrunkImm, ExpandedImm](unsigned CandidateImm) -> bool {
20159 return (ShrunkImm & CandidateImm) == ShrunkImm &&
20160 (~ExpandedImm & CandidateImm) == 0;
20161 };
20162 auto UseImm = [Imm, Opc, Op, VT, &TLO](unsigned NewImm) -> bool {
20163 if (NewImm == Imm)
20164 return true;
20165 SDLoc DL(Op);
20166 SDValue NewC = TLO.DAG.getConstant(NewImm, DL, VT);
20167 SDValue NewOp =
20168 TLO.DAG.getNode(Opc, DL, VT, Op.getOperand(0), NewC, Op->getFlags());
20169 return TLO.CombineTo(Op, NewOp);
20170 };
20171
20172 // Shrunk immediate is 0: AND becomes zero; OR/XOR with 0 leaves the other
20173 // operand (still valid on demanded bits).
20174 if (ShrunkImm == 0) {
20175 ++NumOptimizedImms;
20176 return UseImm(ShrunkImm);
20177 }
20178
20179 // If the immediate is all ones: for AND this removes the operation; for
20180 // OR/XOR it remains a transform valid on demanded bits. (Target-independent
20181 // shrink may not fold this, so keep it to avoid obscure combine loops.)
20182 if (ExpandedImm == ~0U) {
20183 ++NumOptimizedImms;
20184 return UseImm(ExpandedImm);
20185 }
20186
20187 // Thumb1: prefer 0xFF / 0xFFFF when they fit the demanded-bit envelope so
20188 // lowering can match uxtb / uxth (AND immediates only; OR/XOR do not use
20189 // that). Run this before strict ShrunkImm: a tight 8-bit ShrunkImm can be
20190 // legal while 0xFF still matches the envelope and yields better isel (uxtb).
20191 if (Opc == ISD::AND && Subtarget->hasV6Ops()) {
20192 if (IsLegalImm(0xFF)) {
20193 ++NumOptimizedImms;
20194 return UseImm(0xFF);
20195 }
20196
20197 if (IsLegalImm(0xFFFF)) {
20198 ++NumOptimizedImms;
20199 return UseImm(0xFFFF);
20200 }
20201 }
20202
20203 // Don't optimize if it is legal.
20204 if (isLegalLogicalImmediate(Imm, Subtarget))
20205 return false;
20206
20207 // FIXME: Check for BIC being legal causes infinite loop due to target
20208 // independent DAG combine undoing this.
20209
20210 // Prefer strict shrink when ShrunkImm encodes for this target, before
20211 // complement expansion.
20212 if (isLegalLogicalImmediate(ShrunkImm, Subtarget)) {
20213 ++NumOptimizedImms;
20214 return UseImm(ShrunkImm);
20215 }
20216
20217 // Complement expansion: if all undemanded bits are already one, ExpandedImm
20218 // is Imm with every non-demanded bit set. When (~ExpandedImm) < 256, the
20219 // complement fits in an 8-bit unsigned value, i.e. bits 8–31 of ExpandedImm
20220 // are all ones; only the low byte may differ from ~0. Use that expanded
20221 // constant so isel sees a mask shape that fits logical-immediate patterns.
20222 if ((~ExpandedImm) < 256) {
20223 ++NumOptimizedImms;
20224 return UseImm(ExpandedImm);
20225 }
20226
20227 // FIXME: The check for v6 is because this interferes with some ubfx
20228 // optimizations.
20229 if (Opc == ISD::AND && isLegalLogicalImmediate(~ExpandedImm, Subtarget) &&
20230 !Subtarget->hasV6Ops()) {
20231 ++NumOptimizedImms;
20232 return UseImm(ExpandedImm);
20233 }
20234
20235 // Potential improvements:
20236 //
20237 // We could try to recognize lsls+lsrs or lsrs+lsls pairs here.
20238 // We could try to prefer Thumb1 immediates which can be lowered to a
20239 // two-instruction sequence.
20240
20241 return false;
20242}
20243
20245 SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts,
20246 TargetLoweringOpt &TLO) const {
20247 // Delay this optimization to as late as possible.
20248 if (!TLO.LegalOps)
20249 return false;
20250
20251 EVT VT = Op.getValueType();
20252
20253 // Ignore vectors.
20254 if (VT.isVector())
20255 return false;
20256
20257 unsigned Size = VT.getSizeInBits();
20258
20259 if (Size != 32)
20260 return false;
20261
20262 // Exit early if we demand all bits.
20263 if (DemandedBits.isAllOnes())
20264 return false;
20265
20266 switch (Op.getOpcode()) {
20267 default:
20268 return false;
20269 case ISD::AND:
20270 case ISD::OR:
20271 case ISD::XOR:
20272 break;
20273 }
20274 ConstantSDNode *C = dyn_cast<ConstantSDNode>(Op.getOperand(1));
20275 if (!C)
20276 return false;
20277 unsigned Imm = C->getZExtValue();
20278 return optimizeLogicalImm(Op, Imm, DemandedBits, Subtarget, TLO);
20279}
20280
20282 SDValue Op, const APInt &OriginalDemandedBits,
20283 const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO,
20284 unsigned Depth) const {
20285 unsigned Opc = Op.getOpcode();
20286
20287 switch (Opc) {
20288 case ARMISD::ASRL:
20289 case ARMISD::LSRL: {
20290 // If this is result 0 and the other result is unused, see if the demand
20291 // bits allow us to shrink this long shift into a standard small shift in
20292 // the opposite direction.
20293 if (Op.getResNo() == 0 && !Op->hasAnyUseOfValue(1) &&
20294 isa<ConstantSDNode>(Op->getOperand(2))) {
20295 unsigned ShAmt = Op->getConstantOperandVal(2);
20296 if (ShAmt < 32 && OriginalDemandedBits.isSubsetOf(APInt::getAllOnes(32)
20297 << (32 - ShAmt)))
20298 return TLO.CombineTo(
20299 Op, TLO.DAG.getNode(
20300 ISD::SHL, SDLoc(Op), MVT::i32, Op.getOperand(1),
20301 TLO.DAG.getConstant(32 - ShAmt, SDLoc(Op), MVT::i32)));
20302 }
20303 break;
20304 }
20305 case ARMISD::VBICIMM: {
20306 SDValue Op0 = Op.getOperand(0);
20307 unsigned ModImm = Op.getConstantOperandVal(1);
20308 unsigned EltBits = 0;
20309 uint64_t Mask = ARM_AM::decodeVMOVModImm(ModImm, EltBits);
20310 if ((OriginalDemandedBits & Mask) == 0)
20311 return TLO.CombineTo(Op, Op0);
20312 }
20313 }
20314
20316 Op, OriginalDemandedBits, OriginalDemandedElts, Known, TLO, Depth);
20317}
20318
20319//===----------------------------------------------------------------------===//
20320// ARM Inline Assembly Support
20321//===----------------------------------------------------------------------===//
20322
20323const char *ARMTargetLowering::LowerXConstraint(EVT ConstraintVT) const {
20324 // At this point, we have to lower this constraint to something else, so we
20325 // lower it to an "r" or "w". However, by doing this we will force the result
20326 // to be in register, while the X constraint is much more permissive.
20327 //
20328 // Although we are correct (we are free to emit anything, without
20329 // constraints), we might break use cases that would expect us to be more
20330 // efficient and emit something else.
20331 if (!Subtarget->hasVFP2Base())
20332 return "r";
20333 if (ConstraintVT.isFloatingPoint())
20334 return "w";
20335 if (ConstraintVT.isVector() && Subtarget->hasNEON() &&
20336 (ConstraintVT.getSizeInBits() == 64 ||
20337 ConstraintVT.getSizeInBits() == 128))
20338 return "w";
20339
20340 return "r";
20341}
20342
20343/// getConstraintType - Given a constraint letter, return the type of
20344/// constraint it is for this target.
20347 unsigned S = Constraint.size();
20348 if (S == 1) {
20349 switch (Constraint[0]) {
20350 default: break;
20351 case 'l': return C_RegisterClass;
20352 case 'w': return C_RegisterClass;
20353 case 'h': return C_RegisterClass;
20354 case 'x': return C_RegisterClass;
20355 case 't': return C_RegisterClass;
20356 case 'j': return C_Immediate; // Constant for movw.
20357 // An address with a single base register. Due to the way we
20358 // currently handle addresses it is the same as an 'r' memory constraint.
20359 case 'Q': return C_Memory;
20360 }
20361 } else if (S == 2) {
20362 switch (Constraint[0]) {
20363 default: break;
20364 case 'T': return C_RegisterClass;
20365 // All 'U+' constraints are addresses.
20366 case 'U': return C_Memory;
20367 }
20368 }
20369 return TargetLowering::getConstraintType(Constraint);
20370}
20371
20372/// Examine constraint type and operand type and determine a weight value.
20373/// This object must already have been set up with the operand type
20374/// and the current alternative constraint selected.
20377 AsmOperandInfo &info, const char *constraint) const {
20379 Value *CallOperandVal = info.CallOperandVal;
20380 // If we don't have a value, we can't do a match,
20381 // but allow it at the lowest weight.
20382 if (!CallOperandVal)
20383 return CW_Default;
20384 Type *type = CallOperandVal->getType();
20385 // Look at the constraint type.
20386 switch (*constraint) {
20387 default:
20389 break;
20390 case 'l':
20391 if (type->isIntegerTy()) {
20392 if (Subtarget->isThumb())
20393 weight = CW_SpecificReg;
20394 else
20395 weight = CW_Register;
20396 }
20397 break;
20398 case 'w':
20399 if (type->isFloatingPointTy())
20400 weight = CW_Register;
20401 break;
20402 }
20403 return weight;
20404}
20405
20406static bool isIncompatibleReg(const MCPhysReg &PR, MVT VT) {
20407 if (PR == 0 || VT == MVT::Other)
20408 return false;
20409 if (ARM::SPRRegClass.contains(PR))
20410 return VT != MVT::f32 && VT != MVT::f16 && VT != MVT::i32;
20411 if (ARM::DPRRegClass.contains(PR))
20412 return VT != MVT::f64 && !VT.is64BitVector();
20413 return false;
20414}
20415
20416using RCPair = std::pair<unsigned, const TargetRegisterClass *>;
20417
20419 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
20420 switch (Constraint.size()) {
20421 case 1:
20422 // GCC ARM Constraint Letters
20423 switch (Constraint[0]) {
20424 case 'l': // Low regs or general regs.
20425 if (Subtarget->isThumb())
20426 return RCPair(0U, &ARM::tGPRRegClass);
20427 return RCPair(0U, &ARM::GPRRegClass);
20428 case 'h': // High regs or no regs.
20429 if (Subtarget->isThumb())
20430 return RCPair(0U, &ARM::hGPRRegClass);
20431 break;
20432 case 'r':
20433 if (Subtarget->isThumb1Only())
20434 return RCPair(0U, &ARM::tGPRRegClass);
20435 return RCPair(0U, &ARM::GPRRegClass);
20436 case 'w':
20437 if (VT == MVT::Other)
20438 break;
20439 if (VT == MVT::f32 || VT == MVT::f16 || VT == MVT::bf16)
20440 return RCPair(0U, &ARM::SPRRegClass);
20441 if (VT.getSizeInBits() == 64)
20442 return RCPair(0U, &ARM::DPRRegClass);
20443 if (VT.getSizeInBits() == 128)
20444 return RCPair(0U, &ARM::QPRRegClass);
20445 break;
20446 case 'x':
20447 if (VT == MVT::Other)
20448 break;
20449 if (VT == MVT::f32 || VT == MVT::f16 || VT == MVT::bf16)
20450 return RCPair(0U, &ARM::SPR_8RegClass);
20451 if (VT.getSizeInBits() == 64)
20452 return RCPair(0U, &ARM::DPR_8RegClass);
20453 if (VT.getSizeInBits() == 128)
20454 return RCPair(0U, &ARM::QPR_8RegClass);
20455 break;
20456 case 't':
20457 if (VT == MVT::Other)
20458 break;
20459 if (VT == MVT::f32 || VT == MVT::i32 || VT == MVT::f16 || VT == MVT::bf16)
20460 return RCPair(0U, &ARM::SPRRegClass);
20461 if (VT.getSizeInBits() == 64)
20462 return RCPair(0U, &ARM::DPR_VFP2RegClass);
20463 if (VT.getSizeInBits() == 128)
20464 return RCPair(0U, &ARM::QPR_VFP2RegClass);
20465 break;
20466 }
20467 break;
20468
20469 case 2:
20470 if (Constraint[0] == 'T') {
20471 switch (Constraint[1]) {
20472 default:
20473 break;
20474 case 'e':
20475 return RCPair(0U, &ARM::tGPREvenRegClass);
20476 case 'o':
20477 return RCPair(0U, &ARM::tGPROddRegClass);
20478 }
20479 }
20480 break;
20481
20482 default:
20483 break;
20484 }
20485
20486 if (StringRef("{cc}").equals_insensitive(Constraint))
20487 return std::make_pair(unsigned(ARM::CPSR), &ARM::CCRRegClass);
20488
20489 // r14 is an alias of lr.
20490 if (StringRef("{r14}").equals_insensitive(Constraint))
20491 Constraint = "{lr}";
20492
20493 auto RCP = TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
20494 if (isIncompatibleReg(RCP.first, VT))
20495 return {0, nullptr};
20496 return RCP;
20497}
20498
20499/// LowerAsmOperandForConstraint - Lower the specified operand into the Ops
20500/// vector. If it is invalid, don't add anything to Ops.
20502 StringRef Constraint,
20503 std::vector<SDValue> &Ops,
20504 SelectionDAG &DAG) const {
20505 SDValue Result;
20506
20507 // Currently only support length 1 constraints.
20508 if (Constraint.size() != 1)
20509 return;
20510
20511 char ConstraintLetter = Constraint[0];
20512 switch (ConstraintLetter) {
20513 default: break;
20514 case 'j':
20515 case 'I': case 'J': case 'K': case 'L':
20516 case 'M': case 'N': case 'O':
20518 if (!C)
20519 return;
20520
20521 int64_t CVal64 = C->getSExtValue();
20522 int CVal = (int) CVal64;
20523 // None of these constraints allow values larger than 32 bits. Check
20524 // that the value fits in an int.
20525 if (CVal != CVal64)
20526 return;
20527
20528 switch (ConstraintLetter) {
20529 case 'j':
20530 // Constant suitable for movw, must be between 0 and
20531 // 65535.
20532 if (Subtarget->hasV6T2Ops() || (Subtarget->hasV8MBaselineOps()))
20533 if (CVal >= 0 && CVal <= 65535)
20534 break;
20535 return;
20536 case 'I':
20537 if (Subtarget->isThumb1Only()) {
20538 // This must be a constant between 0 and 255, for ADD
20539 // immediates.
20540 if (CVal >= 0 && CVal <= 255)
20541 break;
20542 } else if (Subtarget->isThumb2()) {
20543 // A constant that can be used as an immediate value in a
20544 // data-processing instruction.
20545 if (ARM_AM::getT2SOImmVal(CVal) != -1)
20546 break;
20547 } else {
20548 // A constant that can be used as an immediate value in a
20549 // data-processing instruction.
20550 if (ARM_AM::getSOImmVal(CVal) != -1)
20551 break;
20552 }
20553 return;
20554
20555 case 'J':
20556 if (Subtarget->isThumb1Only()) {
20557 // This must be a constant between -255 and -1, for negated ADD
20558 // immediates. This can be used in GCC with an "n" modifier that
20559 // prints the negated value, for use with SUB instructions. It is
20560 // not useful otherwise but is implemented for compatibility.
20561 if (CVal >= -255 && CVal <= -1)
20562 break;
20563 } else {
20564 // This must be a constant between -4095 and 4095. This is suitable
20565 // for use as the immediate offset field in LDR and STR instructions
20566 // such as LDR r0,[r1,#offset].
20567 if (CVal >= -4095 && CVal <= 4095)
20568 break;
20569 }
20570 return;
20571
20572 case 'K':
20573 if (Subtarget->isThumb1Only()) {
20574 // A 32-bit value where only one byte has a nonzero value. Exclude
20575 // zero to match GCC. This constraint is used by GCC internally for
20576 // constants that can be loaded with a move/shift combination.
20577 // It is not useful otherwise but is implemented for compatibility.
20578 if (CVal != 0 && ARM_AM::isThumbImmShiftedVal(CVal))
20579 break;
20580 } else if (Subtarget->isThumb2()) {
20581 // A constant whose bitwise inverse can be used as an immediate
20582 // value in a data-processing instruction. This can be used in GCC
20583 // with a "B" modifier that prints the inverted value, for use with
20584 // BIC and MVN instructions. It is not useful otherwise but is
20585 // implemented for compatibility.
20586 if (ARM_AM::getT2SOImmVal(~CVal) != -1)
20587 break;
20588 } else {
20589 // A constant whose bitwise inverse can be used as an immediate
20590 // value in a data-processing instruction. This can be used in GCC
20591 // with a "B" modifier that prints the inverted value, for use with
20592 // BIC and MVN instructions. It is not useful otherwise but is
20593 // implemented for compatibility.
20594 if (ARM_AM::getSOImmVal(~CVal) != -1)
20595 break;
20596 }
20597 return;
20598
20599 case 'L':
20600 if (Subtarget->isThumb1Only()) {
20601 // This must be a constant between -7 and 7,
20602 // for 3-operand ADD/SUB immediate instructions.
20603 if (CVal >= -7 && CVal < 7)
20604 break;
20605 } else if (Subtarget->isThumb2()) {
20606 // A constant whose negation can be used as an immediate value in a
20607 // data-processing instruction. This can be used in GCC with an "n"
20608 // modifier that prints the negated value, for use with SUB
20609 // instructions. It is not useful otherwise but is implemented for
20610 // compatibility.
20611 if (ARM_AM::getT2SOImmVal(-CVal) != -1)
20612 break;
20613 } else {
20614 // A constant whose negation can be used as an immediate value in a
20615 // data-processing instruction. This can be used in GCC with an "n"
20616 // modifier that prints the negated value, for use with SUB
20617 // instructions. It is not useful otherwise but is implemented for
20618 // compatibility.
20619 if (ARM_AM::getSOImmVal(-CVal) != -1)
20620 break;
20621 }
20622 return;
20623
20624 case 'M':
20625 if (Subtarget->isThumb1Only()) {
20626 // This must be a multiple of 4 between 0 and 1020, for
20627 // ADD sp + immediate.
20628 if ((CVal >= 0 && CVal <= 1020) && ((CVal & 3) == 0))
20629 break;
20630 } else {
20631 // A power of two or a constant between 0 and 32. This is used in
20632 // GCC for the shift amount on shifted register operands, but it is
20633 // useful in general for any shift amounts.
20634 if ((CVal >= 0 && CVal <= 32) || ((CVal & (CVal - 1)) == 0))
20635 break;
20636 }
20637 return;
20638
20639 case 'N':
20640 if (Subtarget->isThumb1Only()) {
20641 // This must be a constant between 0 and 31, for shift amounts.
20642 if (CVal >= 0 && CVal <= 31)
20643 break;
20644 }
20645 return;
20646
20647 case 'O':
20648 if (Subtarget->isThumb1Only()) {
20649 // This must be a multiple of 4 between -508 and 508, for
20650 // ADD/SUB sp = sp + immediate.
20651 if ((CVal >= -508 && CVal <= 508) && ((CVal & 3) == 0))
20652 break;
20653 }
20654 return;
20655 }
20656 Result = DAG.getSignedTargetConstant(CVal, SDLoc(Op), Op.getValueType());
20657 break;
20658 }
20659
20660 if (Result.getNode()) {
20661 Ops.push_back(Result);
20662 return;
20663 }
20664 return TargetLowering::LowerAsmOperandForConstraint(Op, Constraint, Ops, DAG);
20665}
20666
20667static RTLIB::Libcall getDivRemLibcall(
20668 const SDNode *N, MVT::SimpleValueType SVT) {
20669 assert((N->getOpcode() == ISD::SDIVREM || N->getOpcode() == ISD::UDIVREM ||
20670 N->getOpcode() == ISD::SREM || N->getOpcode() == ISD::UREM) &&
20671 "Unhandled Opcode in getDivRemLibcall");
20672 bool isSigned = N->getOpcode() == ISD::SDIVREM ||
20673 N->getOpcode() == ISD::SREM;
20674 RTLIB::Libcall LC;
20675 switch (SVT) {
20676 default: llvm_unreachable("Unexpected request for libcall!");
20677 case MVT::i8: LC = isSigned ? RTLIB::SDIVREM_I8 : RTLIB::UDIVREM_I8; break;
20678 case MVT::i16: LC = isSigned ? RTLIB::SDIVREM_I16 : RTLIB::UDIVREM_I16; break;
20679 case MVT::i32: LC = isSigned ? RTLIB::SDIVREM_I32 : RTLIB::UDIVREM_I32; break;
20680 case MVT::i64: LC = isSigned ? RTLIB::SDIVREM_I64 : RTLIB::UDIVREM_I64; break;
20681 }
20682 return LC;
20683}
20684
20687 const AttributeList &FuncAttrs, RTLIB::LibcallImpl LCImpl) {
20688 assert((N->getOpcode() == ISD::SDIVREM || N->getOpcode() == ISD::UDIVREM ||
20689 N->getOpcode() == ISD::SREM || N->getOpcode() == ISD::UREM) &&
20690 "Unhandled Opcode in getDivRemArgList");
20691 SDValue Ops[2] = {N->getOperand(0), N->getOperand(1)};
20692
20693 // The Windows __rt_*div* helpers take the divisor before the dividend.
20694 switch (LCImpl) {
20695 case RTLIB::impl___rt_sdiv:
20696 case RTLIB::impl___rt_udiv:
20697 case RTLIB::impl___rt_sdiv64:
20698 case RTLIB::impl___rt_udiv64:
20699 std::swap(Ops[0], Ops[1]);
20700 break;
20701 default:
20702 break;
20703 }
20704
20705 return TargetLowering::getArgListForFunctionType(FuncTy, FuncAttrs, Ops);
20706}
20707
20708SDValue ARMTargetLowering::LowerDivRem(SDValue Op, SelectionDAG &DAG) const {
20709 assert((Subtarget->isTargetAEABI() || Subtarget->isTargetAndroid() ||
20710 Subtarget->isTargetGNUAEABI() || Subtarget->isTargetMuslAEABI() ||
20711 Subtarget->isTargetFuchsia() || Subtarget->isTargetWindows()) &&
20712 "Register-based DivRem lowering only");
20713 unsigned Opcode = Op->getOpcode();
20714 assert((Opcode == ISD::SDIVREM || Opcode == ISD::UDIVREM) &&
20715 "Invalid opcode for Div/Rem lowering");
20716 bool isSigned = (Opcode == ISD::SDIVREM);
20717 EVT VT = Op->getValueType(0);
20718 SDLoc dl(Op);
20719
20720 if (VT == MVT::i64 && isa<ConstantSDNode>(Op.getOperand(1))) {
20722 if (expandDIVREMByConstant(Op.getNode(), Result, MVT::i32, DAG)) {
20723 SDValue Res0 =
20724 DAG.getNode(ISD::BUILD_PAIR, dl, VT, Result[0], Result[1]);
20725 SDValue Res1 =
20726 DAG.getNode(ISD::BUILD_PAIR, dl, VT, Result[2], Result[3]);
20727 return DAG.getNode(ISD::MERGE_VALUES, dl, Op->getVTList(),
20728 {Res0, Res1});
20729 }
20730 }
20731
20732 // If the target has hardware divide, use divide + multiply + subtract:
20733 // div = a / b
20734 // rem = a - b * div
20735 // return {div, rem}
20736 // This should be lowered into UDIV/SDIV + MLS later on.
20737 bool hasDivide = Subtarget->isThumb() ? Subtarget->hasDivideInThumbMode()
20738 : Subtarget->hasDivideInARMMode();
20739 if (hasDivide && Op->getValueType(0).isSimple() &&
20740 Op->getSimpleValueType(0) == MVT::i32) {
20741 unsigned DivOpcode = isSigned ? ISD::SDIV : ISD::UDIV;
20742 const SDValue Dividend = Op->getOperand(0);
20743 const SDValue Divisor = Op->getOperand(1);
20744 SDValue Div = DAG.getNode(DivOpcode, dl, VT, Dividend, Divisor);
20745 SDValue Mul = DAG.getNode(ISD::MUL, dl, VT, Div, Divisor);
20746 SDValue Rem = DAG.getNode(ISD::SUB, dl, VT, Dividend, Mul);
20747
20748 SDValue Values[2] = {Div, Rem};
20749 return DAG.getNode(ISD::MERGE_VALUES, dl, DAG.getVTList(VT, VT), Values);
20750 }
20751
20752 RTLIB::Libcall LC = getDivRemLibcall(Op.getNode(),
20753 VT.getSimpleVT().SimpleTy);
20754 RTLIB::LibcallImpl LCImpl = DAG.getLibcalls().getLibcallImpl(LC);
20755 if (LCImpl == RTLIB::Unsupported)
20756 return SDValue();
20757
20758 auto [FuncTy, FuncAttrs] =
20760 *DAG.getContext(), getTM().getTargetTriple(), DAG.getDataLayout(),
20761 LCImpl);
20762 Type *RetTy = FuncTy->getReturnType();
20763
20764 SDValue InChain = DAG.getEntryNode();
20765
20767 getDivRemArgList(Op.getNode(), FuncTy, FuncAttrs, LCImpl);
20768
20769 SDValue Callee =
20770 DAG.getExternalSymbol(LCImpl, getPointerTy(DAG.getDataLayout()));
20771
20772 if (getTM().getTargetTriple().isOSWindows())
20773 InChain = WinDBZCheckDenominator(DAG, Op.getNode(), InChain);
20774
20775 TargetLowering::CallLoweringInfo CLI(DAG);
20776 CLI.setDebugLoc(dl)
20777 .setChain(InChain)
20778 .setCallee(DAG.getLibcalls().getLibcallImplCallingConv(LCImpl), RetTy,
20779 Callee, std::move(Args))
20780 .setInRegister()
20783
20784 std::pair<SDValue, SDValue> CallInfo = LowerCallTo(CLI);
20785 return CallInfo.first;
20786}
20787
20788// Lowers REM using divmod helpers
20789// see RTABI section 4.2/4.3
20790SDValue ARMTargetLowering::LowerREM(SDNode *N, SelectionDAG &DAG) const {
20791 EVT VT = N->getValueType(0);
20792
20793 if (VT == MVT::i64 && isa<ConstantSDNode>(N->getOperand(1))) {
20795 if (expandDIVREMByConstant(N, Result, MVT::i32, DAG))
20796 return DAG.getNode(ISD::BUILD_PAIR, SDLoc(N), N->getValueType(0),
20797 Result[0], Result[1]);
20798 }
20799
20800 RTLIB::Libcall LC = getDivRemLibcall(N, N->getValueType(0).getSimpleVT().
20801 SimpleTy);
20802 RTLIB::LibcallImpl LCImpl = DAG.getLibcalls().getLibcallImpl(LC);
20803 if (LCImpl == RTLIB::Unsupported)
20804 return SDValue();
20805
20806 auto [FuncTy, FuncAttrs] =
20808 *DAG.getContext(), getTM().getTargetTriple(), DAG.getDataLayout(),
20809 LCImpl);
20810 Type *RetTy = FuncTy->getReturnType();
20811
20812 SDValue InChain = DAG.getEntryNode();
20814 getDivRemArgList(N, FuncTy, FuncAttrs, LCImpl);
20815 bool isSigned = N->getOpcode() == ISD::SREM;
20816
20817 SDValue Callee =
20818 DAG.getExternalSymbol(LCImpl, getPointerTy(DAG.getDataLayout()));
20819
20820 if (getTM().getTargetTriple().isOSWindows())
20821 InChain = WinDBZCheckDenominator(DAG, N, InChain);
20822
20823 // Lower call
20824 CallLoweringInfo CLI(DAG);
20825 CLI.setChain(InChain)
20826 .setCallee(DAG.getLibcalls().getLibcallImplCallingConv(LCImpl), RetTy,
20827 Callee, std::move(Args))
20830 .setDebugLoc(SDLoc(N));
20831 std::pair<SDValue, SDValue> CallResult = LowerCallTo(CLI);
20832
20833 // Return second (rem) result operand (first contains div)
20834 SDNode *ResNode = CallResult.first.getNode();
20835 assert(ResNode->getNumOperands() == 2 && "divmod should return two operands");
20836 return ResNode->getOperand(1);
20837}
20838
20839SDValue
20840ARMTargetLowering::LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const {
20841 assert(getTM().getTargetTriple().isOSWindows() &&
20842 "unsupported target platform");
20843 SDLoc DL(Op);
20844
20845 // Get the inputs.
20846 SDValue Chain = Op.getOperand(0);
20847 SDValue Size = Op.getOperand(1);
20848
20850 "no-stack-arg-probe")) {
20851 MaybeAlign Align =
20852 cast<ConstantSDNode>(Op.getOperand(2))->getMaybeAlignValue();
20853 SDValue SP = DAG.getCopyFromReg(Chain, DL, ARM::SP, MVT::i32);
20854 Chain = SP.getValue(1);
20855 SP = DAG.getNode(ISD::SUB, DL, MVT::i32, SP, Size);
20856 if (Align)
20857 SP = DAG.getNode(ISD::AND, DL, MVT::i32, SP.getValue(0),
20858 DAG.getSignedConstant(-Align->value(), DL, MVT::i32));
20859 Chain = DAG.getCopyToReg(Chain, DL, ARM::SP, SP);
20860 SDValue Ops[2] = { SP, Chain };
20861 return DAG.getMergeValues(Ops, DL);
20862 }
20863
20864 SDValue Words = DAG.getNode(ISD::SRL, DL, MVT::i32, Size,
20865 DAG.getConstant(2, DL, MVT::i32));
20866
20867 SDValue Glue;
20868 Chain = DAG.getCopyToReg(Chain, DL, ARM::R4, Words, Glue);
20869 Glue = Chain.getValue(1);
20870
20871 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
20872 Chain = DAG.getNode(ARMISD::WIN__CHKSTK, DL, NodeTys, Chain, Glue);
20873
20874 SDValue NewSP = DAG.getCopyFromReg(Chain, DL, ARM::SP, MVT::i32);
20875 Chain = NewSP.getValue(1);
20876
20877 SDValue Ops[2] = { NewSP, Chain };
20878 return DAG.getMergeValues(Ops, DL);
20879}
20880
20881SDValue ARMTargetLowering::LowerFP_EXTEND(SDValue Op, SelectionDAG &DAG) const {
20882 bool IsStrict = Op->isStrictFPOpcode();
20883 SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
20884 const unsigned DstSz = Op.getValueType().getSizeInBits();
20885 const unsigned SrcSz = SrcVal.getValueType().getSizeInBits();
20886 assert(DstSz > SrcSz && DstSz <= 64 && SrcSz >= 16 &&
20887 "Unexpected type for custom-lowering FP_EXTEND");
20888
20889 assert((!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) &&
20890 "With both FP DP and 16, any FP conversion is legal!");
20891
20892 assert(!(DstSz == 32 && Subtarget->hasFP16()) &&
20893 "With FP16, 16 to 32 conversion is legal!");
20894
20895 // Converting from 32 -> 64 is valid if we have FP64.
20896 if (SrcSz == 32 && DstSz == 64 && Subtarget->hasFP64()) {
20897 // FIXME: Remove this when we have strict fp instruction selection patterns
20898 if (IsStrict) {
20899 SDLoc Loc(Op);
20900 SDValue Result = DAG.getNode(ISD::FP_EXTEND,
20901 Loc, Op.getValueType(), SrcVal);
20902 return DAG.getMergeValues({Result, Op.getOperand(0)}, Loc);
20903 }
20904 return Op;
20905 }
20906
20907 // Either we are converting from 16 -> 64, without FP16 and/or
20908 // FP.double-precision or without Armv8-fp. So we must do it in two
20909 // steps.
20910 // Or we are converting from 32 -> 64 without fp.double-precision or 16 -> 32
20911 // without FP16. So we must do a function call.
20912 SDLoc Loc(Op);
20913 RTLIB::Libcall LC;
20914 MakeLibCallOptions CallOptions;
20915 SDValue Chain = IsStrict ? Op.getOperand(0) : SDValue();
20916 for (unsigned Sz = SrcSz; Sz <= 32 && Sz < DstSz; Sz *= 2) {
20917 bool Supported = (Sz == 16 ? Subtarget->hasFP16() : Subtarget->hasFP64());
20918 MVT SrcVT = (Sz == 16 ? MVT::f16 : MVT::f32);
20919 MVT DstVT = (Sz == 16 ? MVT::f32 : MVT::f64);
20920 if (Supported) {
20921 if (IsStrict) {
20922 SrcVal = DAG.getNode(ISD::STRICT_FP_EXTEND, Loc,
20923 {DstVT, MVT::Other}, {Chain, SrcVal});
20924 Chain = SrcVal.getValue(1);
20925 } else {
20926 SrcVal = DAG.getNode(ISD::FP_EXTEND, Loc, DstVT, SrcVal);
20927 }
20928 } else {
20929 LC = RTLIB::getFPEXT(SrcVT, DstVT);
20930 assert(LC != RTLIB::UNKNOWN_LIBCALL &&
20931 "Unexpected type for custom-lowering FP_EXTEND");
20932 std::tie(SrcVal, Chain) = makeLibCall(DAG, LC, DstVT, SrcVal, CallOptions,
20933 Loc, Chain);
20934 }
20935 }
20936
20937 return IsStrict ? DAG.getMergeValues({SrcVal, Chain}, Loc) : SrcVal;
20938}
20939
20940SDValue ARMTargetLowering::LowerFP_ROUND(SDValue Op, SelectionDAG &DAG) const {
20941 bool IsStrict = Op->isStrictFPOpcode();
20942
20943 SDValue SrcVal = Op.getOperand(IsStrict ? 1 : 0);
20944 EVT SrcVT = SrcVal.getValueType();
20945 EVT DstVT = Op.getValueType();
20946
20947 if (DstVT == MVT::bf16) {
20948 if (Subtarget->hasBF16() && SrcVT == MVT::f32)
20949 return Op;
20950 return SDValue();
20951 }
20952
20953 const unsigned DstSz = Op.getValueType().getSizeInBits();
20954 const unsigned SrcSz = SrcVT.getSizeInBits();
20955 (void)DstSz;
20956 assert(DstSz < SrcSz && SrcSz <= 64 && DstSz >= 16 &&
20957 "Unexpected type for custom-lowering FP_ROUND");
20958
20959 assert((!Subtarget->hasFP64() || !Subtarget->hasFPARMv8Base()) &&
20960 "With both FP DP and 16, any FP conversion is legal!");
20961
20962 SDLoc Loc(Op);
20963
20964 // Instruction from 32 -> 16 if hasFP16 is valid
20965 if (SrcSz == 32 && Subtarget->hasFP16())
20966 return Op;
20967
20968 // Lib call from 32 -> 16 / 64 -> [32, 16]
20969 RTLIB::Libcall LC = RTLIB::getFPROUND(SrcVT, DstVT);
20970 assert(LC != RTLIB::UNKNOWN_LIBCALL &&
20971 "Unexpected type for custom-lowering FP_ROUND");
20972 MakeLibCallOptions CallOptions;
20973 SDValue Chain = IsStrict ? Op.getOperand(0) : SDValue();
20974 SDValue Result;
20975 std::tie(Result, Chain) = makeLibCall(DAG, LC, DstVT, SrcVal, CallOptions,
20976 Loc, Chain);
20977 return IsStrict ? DAG.getMergeValues({Result, Chain}, Loc) : Result;
20978}
20979
20980bool
20982 // The ARM target isn't yet aware of offsets.
20983 return false;
20984}
20985
20987 if (v == 0xffffffff)
20988 return false;
20989
20990 // there can be 1's on either or both "outsides", all the "inside"
20991 // bits must be 0's
20992 return isShiftedMask_32(~v);
20993}
20994
20995/// isFPImmLegal - Returns true if the target can instruction select the
20996/// specified FP immediate natively. If false, the legalizer will
20997/// materialize the FP immediate as a load from a constant pool.
20999 bool ForCodeSize) const {
21000 if (!Subtarget->hasVFP3Base())
21001 return false;
21002 if (VT == MVT::f16 && Subtarget->hasFullFP16())
21003 return ARM_AM::getFP16Imm(Imm) != -1;
21004 if (VT == MVT::f32 && Subtarget->hasFullFP16() &&
21006 return true;
21007 if (VT == MVT::f32)
21008 return ARM_AM::getFP32Imm(Imm) != -1;
21009 if (VT == MVT::f64 && Subtarget->hasFP64())
21010 return ARM_AM::getFP64Imm(Imm) != -1;
21011 return false;
21012}
21013
21014/// getTgtMemIntrinsic - Represent NEON load and store intrinsics as
21015/// MemIntrinsicNodes. The associated MachineMemOperands record the alignment
21016/// specified in the intrinsic calls.
21019 MachineFunction &MF, unsigned Intrinsic) const {
21020 IntrinsicInfo Info;
21021 switch (Intrinsic) {
21022 case Intrinsic::arm_neon_vld1:
21023 case Intrinsic::arm_neon_vld2:
21024 case Intrinsic::arm_neon_vld3:
21025 case Intrinsic::arm_neon_vld4:
21026 case Intrinsic::arm_neon_vld2lane:
21027 case Intrinsic::arm_neon_vld3lane:
21028 case Intrinsic::arm_neon_vld4lane:
21029 case Intrinsic::arm_neon_vld2dup:
21030 case Intrinsic::arm_neon_vld3dup:
21031 case Intrinsic::arm_neon_vld4dup: {
21032 Info.opc = ISD::INTRINSIC_W_CHAIN;
21033 // Conservatively set memVT to the entire set of vectors loaded.
21034 auto &DL = I.getDataLayout();
21035 uint64_t NumElts = DL.getTypeSizeInBits(I.getType()) / 64;
21036 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts);
21037 Info.ptrVal = I.getArgOperand(0);
21038 Info.offset = 0;
21039 Value *AlignArg = I.getArgOperand(I.arg_size() - 1);
21040 Info.align = cast<ConstantInt>(AlignArg)->getMaybeAlignValue();
21041 // volatile loads with NEON intrinsics not supported
21042 Info.flags = MachineMemOperand::MOLoad;
21043 Infos.push_back(Info);
21044 return;
21045 }
21046 case Intrinsic::arm_neon_vld1x2:
21047 case Intrinsic::arm_neon_vld1x3:
21048 case Intrinsic::arm_neon_vld1x4: {
21049 Info.opc = ISD::INTRINSIC_W_CHAIN;
21050 // Conservatively set memVT to the entire set of vectors loaded.
21051 auto &DL = I.getDataLayout();
21052 uint64_t NumElts = DL.getTypeSizeInBits(I.getType()) / 64;
21053 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts);
21054 Info.ptrVal = I.getArgOperand(I.arg_size() - 1);
21055 Info.offset = 0;
21056 Info.align = I.getParamAlign(I.arg_size() - 1).valueOrOne();
21057 // volatile loads with NEON intrinsics not supported
21058 Info.flags = MachineMemOperand::MOLoad;
21059 Infos.push_back(Info);
21060 return;
21061 }
21062 case Intrinsic::arm_neon_vst1:
21063 case Intrinsic::arm_neon_vst2:
21064 case Intrinsic::arm_neon_vst3:
21065 case Intrinsic::arm_neon_vst4:
21066 case Intrinsic::arm_neon_vst2lane:
21067 case Intrinsic::arm_neon_vst3lane:
21068 case Intrinsic::arm_neon_vst4lane: {
21069 Info.opc = ISD::INTRINSIC_VOID;
21070 // Conservatively set memVT to the entire set of vectors stored.
21071 auto &DL = I.getDataLayout();
21072 unsigned NumElts = 0;
21073 for (unsigned ArgI = 1, ArgE = I.arg_size(); ArgI < ArgE; ++ArgI) {
21074 Type *ArgTy = I.getArgOperand(ArgI)->getType();
21075 if (!ArgTy->isVectorTy())
21076 break;
21077 NumElts += DL.getTypeSizeInBits(ArgTy) / 64;
21078 }
21079 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts);
21080 Info.ptrVal = I.getArgOperand(0);
21081 Info.offset = 0;
21082 Value *AlignArg = I.getArgOperand(I.arg_size() - 1);
21083 Info.align = cast<ConstantInt>(AlignArg)->getMaybeAlignValue();
21084 // volatile stores with NEON intrinsics not supported
21085 Info.flags = MachineMemOperand::MOStore;
21086 Infos.push_back(Info);
21087 return;
21088 }
21089 case Intrinsic::arm_neon_vst1x2:
21090 case Intrinsic::arm_neon_vst1x3:
21091 case Intrinsic::arm_neon_vst1x4: {
21092 Info.opc = ISD::INTRINSIC_VOID;
21093 // Conservatively set memVT to the entire set of vectors stored.
21094 auto &DL = I.getDataLayout();
21095 unsigned NumElts = 0;
21096 for (unsigned ArgI = 1, ArgE = I.arg_size(); ArgI < ArgE; ++ArgI) {
21097 Type *ArgTy = I.getArgOperand(ArgI)->getType();
21098 if (!ArgTy->isVectorTy())
21099 break;
21100 NumElts += DL.getTypeSizeInBits(ArgTy) / 64;
21101 }
21102 Info.memVT = EVT::getVectorVT(I.getType()->getContext(), MVT::i64, NumElts);
21103 Info.ptrVal = I.getArgOperand(0);
21104 Info.offset = 0;
21105 Info.align = I.getParamAlign(0).valueOrOne();
21106 // volatile stores with NEON intrinsics not supported
21107 Info.flags = MachineMemOperand::MOStore;
21108 Infos.push_back(Info);
21109 return;
21110 }
21111 case Intrinsic::arm_mve_vld2q:
21112 case Intrinsic::arm_mve_vld4q: {
21113 Info.opc = ISD::INTRINSIC_W_CHAIN;
21114 // Conservatively set memVT to the entire set of vectors loaded.
21115 Type *VecTy = cast<StructType>(I.getType())->getElementType(1);
21116 unsigned Factor = Intrinsic == Intrinsic::arm_mve_vld2q ? 2 : 4;
21117 Info.memVT = EVT::getVectorVT(VecTy->getContext(), MVT::i64, Factor * 2);
21118 Info.ptrVal = I.getArgOperand(0);
21119 Info.offset = 0;
21120 Info.align = Align(VecTy->getScalarSizeInBits() / 8);
21121 // volatile loads with MVE intrinsics not supported
21122 Info.flags = MachineMemOperand::MOLoad;
21123 Infos.push_back(Info);
21124 return;
21125 }
21126 case Intrinsic::arm_mve_vst2q:
21127 case Intrinsic::arm_mve_vst4q: {
21128 Info.opc = ISD::INTRINSIC_VOID;
21129 // Conservatively set memVT to the entire set of vectors stored.
21130 Type *VecTy = I.getArgOperand(1)->getType();
21131 unsigned Factor = Intrinsic == Intrinsic::arm_mve_vst2q ? 2 : 4;
21132 Info.memVT = EVT::getVectorVT(VecTy->getContext(), MVT::i64, Factor * 2);
21133 Info.ptrVal = I.getArgOperand(0);
21134 Info.offset = 0;
21135 Info.align = Align(VecTy->getScalarSizeInBits() / 8);
21136 // volatile stores with MVE intrinsics not supported
21137 Info.flags = MachineMemOperand::MOStore;
21138 Infos.push_back(Info);
21139 return;
21140 }
21141 case Intrinsic::arm_mve_vldr_gather_base:
21142 case Intrinsic::arm_mve_vldr_gather_base_predicated: {
21143 Info.opc = ISD::INTRINSIC_W_CHAIN;
21144 Info.ptrVal = nullptr;
21145 Info.memVT = MVT::getVT(I.getType());
21146 Info.align = Align(1);
21147 Info.flags |= MachineMemOperand::MOLoad;
21148 Infos.push_back(Info);
21149 return;
21150 }
21151 case Intrinsic::arm_mve_vldr_gather_base_wb:
21152 case Intrinsic::arm_mve_vldr_gather_base_wb_predicated: {
21153 Info.opc = ISD::INTRINSIC_W_CHAIN;
21154 Info.ptrVal = nullptr;
21155 Info.memVT = MVT::getVT(I.getType()->getContainedType(0));
21156 Info.align = Align(1);
21157 Info.flags |= MachineMemOperand::MOLoad;
21158 Infos.push_back(Info);
21159 return;
21160 }
21161 case Intrinsic::arm_mve_vldr_gather_offset:
21162 case Intrinsic::arm_mve_vldr_gather_offset_predicated: {
21163 Info.opc = ISD::INTRINSIC_W_CHAIN;
21164 Info.ptrVal = nullptr;
21165 MVT DataVT = MVT::getVT(I.getType());
21166 unsigned MemSize = cast<ConstantInt>(I.getArgOperand(2))->getZExtValue();
21167 Info.memVT = MVT::getVectorVT(MVT::getIntegerVT(MemSize),
21168 DataVT.getVectorNumElements());
21169 Info.align = Align(1);
21170 Info.flags |= MachineMemOperand::MOLoad;
21171 Infos.push_back(Info);
21172 return;
21173 }
21174 case Intrinsic::arm_mve_vstr_scatter_base:
21175 case Intrinsic::arm_mve_vstr_scatter_base_predicated: {
21176 Info.opc = ISD::INTRINSIC_VOID;
21177 Info.ptrVal = nullptr;
21178 Info.memVT = MVT::getVT(I.getArgOperand(2)->getType());
21179 Info.align = Align(1);
21180 Info.flags |= MachineMemOperand::MOStore;
21181 Infos.push_back(Info);
21182 return;
21183 }
21184 case Intrinsic::arm_mve_vstr_scatter_base_wb:
21185 case Intrinsic::arm_mve_vstr_scatter_base_wb_predicated: {
21186 Info.opc = ISD::INTRINSIC_W_CHAIN;
21187 Info.ptrVal = nullptr;
21188 Info.memVT = MVT::getVT(I.getArgOperand(2)->getType());
21189 Info.align = Align(1);
21190 Info.flags |= MachineMemOperand::MOStore;
21191 Infos.push_back(Info);
21192 return;
21193 }
21194 case Intrinsic::arm_mve_vstr_scatter_offset:
21195 case Intrinsic::arm_mve_vstr_scatter_offset_predicated: {
21196 Info.opc = ISD::INTRINSIC_VOID;
21197 Info.ptrVal = nullptr;
21198 MVT DataVT = MVT::getVT(I.getArgOperand(2)->getType());
21199 unsigned MemSize = cast<ConstantInt>(I.getArgOperand(3))->getZExtValue();
21200 Info.memVT = MVT::getVectorVT(MVT::getIntegerVT(MemSize),
21201 DataVT.getVectorNumElements());
21202 Info.align = Align(1);
21203 Info.flags |= MachineMemOperand::MOStore;
21204 Infos.push_back(Info);
21205 return;
21206 }
21207 case Intrinsic::arm_ldaex:
21208 case Intrinsic::arm_ldrex: {
21209 auto &DL = I.getDataLayout();
21210 Type *ValTy = I.getParamElementType(0);
21211 Info.opc = ISD::INTRINSIC_W_CHAIN;
21212 Info.memVT = MVT::getVT(ValTy);
21213 Info.ptrVal = I.getArgOperand(0);
21214 Info.offset = 0;
21215 Info.align = DL.getABITypeAlign(ValTy);
21217 Infos.push_back(Info);
21218 return;
21219 }
21220 case Intrinsic::arm_stlex:
21221 case Intrinsic::arm_strex: {
21222 auto &DL = I.getDataLayout();
21223 Type *ValTy = I.getParamElementType(1);
21224 Info.opc = ISD::INTRINSIC_W_CHAIN;
21225 Info.memVT = MVT::getVT(ValTy);
21226 Info.ptrVal = I.getArgOperand(1);
21227 Info.offset = 0;
21228 Info.align = DL.getABITypeAlign(ValTy);
21230 Infos.push_back(Info);
21231 return;
21232 }
21233 case Intrinsic::arm_stlexd:
21234 case Intrinsic::arm_strexd:
21235 Info.opc = ISD::INTRINSIC_W_CHAIN;
21236 Info.memVT = MVT::i64;
21237 Info.ptrVal = I.getArgOperand(2);
21238 Info.offset = 0;
21239 Info.align = Align(8);
21241 Infos.push_back(Info);
21242 return;
21243
21244 case Intrinsic::arm_ldaexd:
21245 case Intrinsic::arm_ldrexd:
21246 Info.opc = ISD::INTRINSIC_W_CHAIN;
21247 Info.memVT = MVT::i64;
21248 Info.ptrVal = I.getArgOperand(0);
21249 Info.offset = 0;
21250 Info.align = Align(8);
21252 Infos.push_back(Info);
21253 return;
21254
21255 default:
21256 break;
21257 }
21258}
21259
21260/// Returns true if it is beneficial to convert a load of a constant
21261/// to just the constant itself.
21263 Type *Ty) const {
21264 assert(Ty->isIntegerTy());
21265
21266 unsigned Bits = Ty->getPrimitiveSizeInBits();
21267 if (Bits == 0 || Bits > 32)
21268 return false;
21269 return true;
21270}
21271
21274 unsigned Index) const {
21277
21278 if (Index == 0 || Index == ResVT.getVectorNumElements())
21281}
21282
21284 ARM_MB::MemBOpt Domain) const {
21285 // First, if the target has no DMB, see what fallback we can use.
21286 if (!Subtarget->hasDataBarrier()) {
21287 // Some ARMv6 cpus can support data barriers with an mcr instruction.
21288 // Thumb1 and pre-v6 ARM mode use a libcall instead and should never get
21289 // here.
21290 if (Subtarget->hasV6Ops() && !Subtarget->isThumb()) {
21291 Value* args[6] = {Builder.getInt32(15), Builder.getInt32(0),
21292 Builder.getInt32(0), Builder.getInt32(7),
21293 Builder.getInt32(10), Builder.getInt32(5)};
21294 return Builder.CreateIntrinsicWithoutFolding(Intrinsic::arm_mcr, args);
21295 }
21296 // Instead of using barriers, atomic accesses on these subtargets use
21297 // libcalls.
21298 llvm_unreachable("makeDMB on a target so old that it has no barriers");
21299 } else {
21300 // Only a full system barrier exists in the M-class architectures.
21301 Domain = Subtarget->isMClass() ? ARM_MB::SY : Domain;
21302 Constant *CDomain = Builder.getInt32(Domain);
21303 return Builder.CreateIntrinsicWithoutFolding(Intrinsic::arm_dmb, CDomain);
21304 }
21305}
21306
21307// Based on http://www.cl.cam.ac.uk/~pes20/cpp/cpp0xmappings.html
21309 Instruction *Inst,
21310 AtomicOrdering Ord) const {
21311 switch (Ord) {
21314 llvm_unreachable("Invalid fence: unordered/non-atomic");
21317 return nullptr; // Nothing to do
21319 if (!Inst->hasAtomicStore())
21320 return nullptr; // Nothing to do
21321 [[fallthrough]];
21324 if (Subtarget->preferISHSTBarriers())
21325 return makeDMB(Builder, ARM_MB::ISHST);
21326 // FIXME: add a comment with a link to documentation justifying this.
21327 else
21328 return makeDMB(Builder, ARM_MB::ISH);
21329 }
21330 llvm_unreachable("Unknown fence ordering in emitLeadingFence");
21331}
21332
21334 Instruction *Inst,
21335 AtomicOrdering Ord) const {
21336 switch (Ord) {
21339 llvm_unreachable("Invalid fence: unordered/not-atomic");
21342 return nullptr; // Nothing to do
21346 return makeDMB(Builder, ARM_MB::ISH);
21347 }
21348 llvm_unreachable("Unknown fence ordering in emitTrailingFence");
21349}
21350
21351// Loads and stores less than 64-bits are already atomic; ones above that
21352// are doomed anyway, so defer to the default libcall and blame the OS when
21353// things go wrong. Cortex M doesn't have ldrexd/strexd though, so don't emit
21354// anything for those.
21357 bool has64BitAtomicStore;
21358 if (Subtarget->isMClass())
21359 has64BitAtomicStore = false;
21360 else if (Subtarget->isThumb())
21361 has64BitAtomicStore = Subtarget->hasV7Ops();
21362 else
21363 has64BitAtomicStore = Subtarget->hasV6Ops();
21364
21365 unsigned Size = SI->getValueOperand()->getType()->getPrimitiveSizeInBits();
21366 return Size == 64 && has64BitAtomicStore ? AtomicExpansionKind::Expand
21368}
21369
21370// Loads and stores less than 64-bits are already atomic; ones above that
21371// are doomed anyway, so defer to the default libcall and blame the OS when
21372// things go wrong. Cortex M doesn't have ldrexd/strexd though, so don't emit
21373// anything for those.
21374// FIXME: ldrd and strd are atomic if the CPU has LPAE (e.g. A15 has that
21375// guarantee, see DDI0406C ARM architecture reference manual,
21376// sections A8.8.72-74 LDRD)
21379 bool has64BitAtomicLoad;
21380 if (Subtarget->isMClass())
21381 has64BitAtomicLoad = false;
21382 else if (Subtarget->isThumb())
21383 has64BitAtomicLoad = Subtarget->hasV7Ops();
21384 else
21385 has64BitAtomicLoad = Subtarget->hasV6Ops();
21386
21387 unsigned Size = LI->getType()->getPrimitiveSizeInBits();
21388 return (Size == 64 && has64BitAtomicLoad) ? AtomicExpansionKind::LLOnly
21390}
21391
21392// For the real atomic operations, we have ldrex/strex up to 32 bits,
21393// and up to 64 bits on the non-M profiles
21396 if (AI->isFloatingPointOperation())
21398
21399 unsigned Size = AI->getType()->getPrimitiveSizeInBits();
21400 bool hasAtomicRMW;
21401 if (Subtarget->isMClass())
21402 hasAtomicRMW = Subtarget->hasV8MBaselineOps();
21403 else if (Subtarget->isThumb())
21404 hasAtomicRMW = Subtarget->hasV7Ops();
21405 else
21406 hasAtomicRMW = Subtarget->hasV6Ops();
21407 if (Size <= (Subtarget->isMClass() ? 32U : 64U) && hasAtomicRMW) {
21408 // At -O0, fast-regalloc cannot cope with the live vregs necessary to
21409 // implement atomicrmw without spilling. If the target address is also on
21410 // the stack and close enough to the spill slot, this can lead to a
21411 // situation where the monitor always gets cleared and the atomic operation
21412 // can never succeed. So at -O0 lower this operation to a CAS loop.
21413 if (getTargetMachine().getOptLevel() == CodeGenOptLevel::None)
21416 }
21418}
21419
21420// Similar to shouldExpandAtomicRMWInIR, ldrex/strex can be used up to 32
21421// bits, and up to 64 bits on the non-M profiles.
21424 const AtomicCmpXchgInst *AI) const {
21425 // At -O0, fast-regalloc cannot cope with the live vregs necessary to
21426 // implement cmpxchg without spilling. If the address being exchanged is also
21427 // on the stack and close enough to the spill slot, this can lead to a
21428 // situation where the monitor always gets cleared and the atomic operation
21429 // can never succeed. So at -O0 we need a late-expanded pseudo-inst instead.
21430 unsigned Size = AI->getOperand(1)->getType()->getPrimitiveSizeInBits();
21431 bool HasAtomicCmpXchg;
21432 if (Subtarget->isMClass())
21433 HasAtomicCmpXchg = Subtarget->hasV8MBaselineOps();
21434 else if (Subtarget->isThumb())
21435 HasAtomicCmpXchg = Subtarget->hasV7Ops();
21436 else
21437 HasAtomicCmpXchg = Subtarget->hasV6Ops();
21438 if (getTargetMachine().getOptLevel() != CodeGenOptLevel::None &&
21439 HasAtomicCmpXchg && Size <= (Subtarget->isMClass() ? 32U : 64U))
21442}
21443
21445 const Instruction *I) const {
21446 return InsertFencesForAtomic;
21447}
21448
21450 // ROPI/RWPI are not supported currently.
21451 return !Subtarget->isROPI() && !Subtarget->isRWPI();
21452}
21453
21455 Module &M, const LibcallLoweringInfo &Libcalls) const {
21456 // MSVC CRT provides functionalities for stack protection.
21457 RTLIB::LibcallImpl SecurityCheckCookieLibcall =
21458 Libcalls.getLibcallImpl(RTLIB::SECURITY_CHECK_COOKIE);
21459
21460 RTLIB::LibcallImpl SecurityCookieVar =
21461 Libcalls.getLibcallImpl(RTLIB::STACK_CHECK_GUARD);
21462 if (SecurityCheckCookieLibcall != RTLIB::Unsupported &&
21463 SecurityCookieVar != RTLIB::Unsupported) {
21464 // MSVC CRT has a global variable holding security cookie.
21465 M.getOrInsertGlobal(getLibcallImplName(SecurityCookieVar),
21466 PointerType::getUnqual(M.getContext()));
21467
21468 // MSVC CRT has a function to validate security cookie.
21469 FunctionCallee SecurityCheckCookie =
21470 M.getOrInsertFunction(getLibcallImplName(SecurityCheckCookieLibcall),
21471 Type::getVoidTy(M.getContext()),
21472 PointerType::getUnqual(M.getContext()));
21473 if (Function *F = dyn_cast<Function>(SecurityCheckCookie.getCallee()))
21474 F->addParamAttr(0, Attribute::AttrKind::InReg);
21475 }
21476
21478}
21479
21481 unsigned &Cost) const {
21482 // If we do not have NEON, vector types are not natively supported.
21483 if (!Subtarget->hasNEON())
21484 return false;
21485
21486 // Floating point values and vector values map to the same register file.
21487 // Therefore, although we could do a store extract of a vector type, this is
21488 // better to leave at float as we have more freedom in the addressing mode for
21489 // those.
21490 if (VectorTy->isFPOrFPVectorTy())
21491 return false;
21492
21493 // If the index is unknown at compile time, this is very expensive to lower
21494 // and it is not possible to combine the store with the extract.
21495 if (!isa<ConstantInt>(Idx))
21496 return false;
21497
21498 assert(VectorTy->isVectorTy() && "VectorTy is not a vector type");
21499 unsigned BitWidth = VectorTy->getPrimitiveSizeInBits().getFixedValue();
21500 // We can do a store + vector extract on any vector that fits perfectly in a D
21501 // or Q register.
21502 if (BitWidth == 64 || BitWidth == 128) {
21503 Cost = 0;
21504 return true;
21505 }
21506 return false;
21507}
21508
21510 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
21511 UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
21512 unsigned Opcode = Op.getOpcode();
21513 switch (Opcode) {
21514 case ARMISD::VORRIMM:
21515 case ARMISD::VBICIMM:
21516 return false;
21517 }
21519 Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
21520}
21521
21523 return Subtarget->hasV5TOps() && !Subtarget->isThumb1Only();
21524}
21525
21527 return Subtarget->hasV5TOps() && !Subtarget->isThumb1Only();
21528}
21529
21531 const Instruction &AndI) const {
21532 if (!Subtarget->hasV7Ops())
21533 return false;
21534
21535 // Sink the `and` instruction only if the mask would fit into a modified
21536 // immediate operand.
21538 if (!Mask || Mask->getValue().getBitWidth() > 32u)
21539 return false;
21540 auto MaskVal = unsigned(Mask->getValue().getZExtValue());
21541 return (Subtarget->isThumb2() ? ARM_AM::getT2SOImmVal(MaskVal)
21542 : ARM_AM::getSOImmVal(MaskVal)) != -1;
21543}
21544
21547 SelectionDAG &DAG, SDNode *N, unsigned ExpansionFactor) const {
21548 if (Subtarget->hasMinSize() && !getTM().getTargetTriple().isOSWindows())
21551 ExpansionFactor);
21552}
21553
21555 Value *Addr,
21556 AtomicOrdering Ord) const {
21557 Module *M = Builder.getModule();
21558 bool IsAcquire = isAcquireOrStronger(Ord);
21559
21560 // Since i64 isn't legal and intrinsics don't get type-lowered, the ldrexd
21561 // intrinsic must return {i32, i32} and we have to recombine them into a
21562 // single i64 here.
21563 if (ValueTy->getPrimitiveSizeInBits() == 64) {
21565 IsAcquire ? Intrinsic::arm_ldaexd : Intrinsic::arm_ldrexd;
21566
21567 Value *LoHi =
21568 Builder.CreateIntrinsic(Int, Addr, /*FMFSource=*/nullptr, "lohi");
21569
21570 Value *Lo = Builder.CreateExtractValue(LoHi, 0, "lo");
21571 Value *Hi = Builder.CreateExtractValue(LoHi, 1, "hi");
21572 if (!Subtarget->isLittle())
21573 std::swap (Lo, Hi);
21574 Lo = Builder.CreateZExt(Lo, ValueTy, "lo64");
21575 Hi = Builder.CreateZExt(Hi, ValueTy, "hi64");
21576 return Builder.CreateOr(
21577 Lo, Builder.CreateShl(Hi, ConstantInt::get(ValueTy, 32)), "val64");
21578 }
21579
21580 Type *Tys[] = { Addr->getType() };
21581 Intrinsic::ID Int = IsAcquire ? Intrinsic::arm_ldaex : Intrinsic::arm_ldrex;
21582 CallInst *CI = Builder.CreateIntrinsicWithoutFolding(Int, Tys, Addr);
21583
21584 CI->addParamAttr(
21585 0, Attribute::get(M->getContext(), Attribute::ElementType, ValueTy));
21586 return Builder.CreateTruncOrBitCast(CI, ValueTy);
21587}
21588
21590 IRBuilderBase &Builder) const {
21591 if (!Subtarget->hasV7Ops())
21592 return;
21593 Builder.CreateIntrinsic(Intrinsic::arm_clrex, {});
21594}
21595
21597 Value *Val, Value *Addr,
21598 AtomicOrdering Ord) const {
21599 Module *M = Builder.getModule();
21600 bool IsRelease = isReleaseOrStronger(Ord);
21601
21602 // Since the intrinsics must have legal type, the i64 intrinsics take two
21603 // parameters: "i32, i32". We must marshal Val into the appropriate form
21604 // before the call.
21605 if (Val->getType()->getPrimitiveSizeInBits() == 64) {
21607 IsRelease ? Intrinsic::arm_stlexd : Intrinsic::arm_strexd;
21608 Type *Int32Ty = Type::getInt32Ty(M->getContext());
21609
21610 Value *Lo = Builder.CreateTrunc(Val, Int32Ty, "lo");
21611 Value *Hi = Builder.CreateTrunc(Builder.CreateLShr(Val, 32), Int32Ty, "hi");
21612 if (!Subtarget->isLittle())
21613 std::swap(Lo, Hi);
21614 return Builder.CreateIntrinsic(Int, {Lo, Hi, Addr});
21615 }
21616
21617 Intrinsic::ID Int = IsRelease ? Intrinsic::arm_stlex : Intrinsic::arm_strex;
21618 Type *Tys[] = { Addr->getType() };
21620
21621 CallInst *CI = Builder.CreateCall(
21622 Strex, {Builder.CreateZExtOrBitCast(
21623 Val, Strex->getFunctionType()->getParamType(0)),
21624 Addr});
21625 CI->addParamAttr(1, Attribute::get(M->getContext(), Attribute::ElementType,
21626 Val->getType()));
21627 return CI;
21628}
21629
21630
21632 return Subtarget->isMClass();
21633}
21634
21635/// A helper function for determining the number of interleaved accesses we
21636/// will generate when lowering accesses of the given type.
21637unsigned
21639 const DataLayout &DL) const {
21640 return (DL.getTypeSizeInBits(VecTy) + 127) / 128;
21641}
21642
21644 unsigned Factor, FixedVectorType *VecTy, Align Alignment,
21645 const DataLayout &DL) const {
21646
21647 unsigned VecSize = DL.getTypeSizeInBits(VecTy);
21648 unsigned ElSize = DL.getTypeSizeInBits(VecTy->getElementType());
21649
21650 if (!Subtarget->hasNEON() && !Subtarget->hasMVEIntegerOps())
21651 return false;
21652
21653 // Ensure the vector doesn't have f16 elements. Even though we could do an
21654 // i16 vldN, we can't hold the f16 vectors and will end up converting via
21655 // f32.
21656 if (Subtarget->hasNEON() && VecTy->getElementType()->isHalfTy())
21657 return false;
21658 if (Subtarget->hasMVEIntegerOps() && Factor == 3)
21659 return false;
21660
21661 // Ensure the number of vector elements is greater than 1.
21662 if (VecTy->getNumElements() < 2)
21663 return false;
21664
21665 // Ensure the element type is legal.
21666 if (ElSize != 8 && ElSize != 16 && ElSize != 32)
21667 return false;
21668 // And the alignment if high enough under MVE.
21669 if (Subtarget->hasMVEIntegerOps() && Alignment < ElSize / 8)
21670 return false;
21671
21672 // Ensure the total vector size is 64 or a multiple of 128. Types larger than
21673 // 128 will be split into multiple interleaved accesses.
21674 if (Subtarget->hasNEON() && VecSize == 64)
21675 return true;
21676 return VecSize % 128 == 0;
21677}
21678
21680 if (Subtarget->hasNEON())
21681 return 4;
21682 if (Subtarget->hasMVEIntegerOps())
21685}
21686
21687/// Lower an interleaved load into a vldN intrinsic.
21688///
21689/// E.g. Lower an interleaved load (Factor = 2):
21690/// %wide.vec = load <8 x i32>, <8 x i32>* %ptr, align 4
21691/// %v0 = shuffle %wide.vec, undef, <0, 2, 4, 6> ; Extract even elements
21692/// %v1 = shuffle %wide.vec, undef, <1, 3, 5, 7> ; Extract odd elements
21693///
21694/// Into:
21695/// %vld2 = { <4 x i32>, <4 x i32> } call llvm.arm.neon.vld2(%ptr, 4)
21696/// %vec0 = extractelement { <4 x i32>, <4 x i32> } %vld2, i32 0
21697/// %vec1 = extractelement { <4 x i32>, <4 x i32> } %vld2, i32 1
21700 ArrayRef<unsigned> Indices, unsigned Factor, const APInt &GapMask) const {
21701 assert(Factor >= 2 && Factor <= getMaxSupportedInterleaveFactor() &&
21702 "Invalid interleave factor");
21703 assert(!Shuffles.empty() && "Empty shufflevector input");
21704 assert(Shuffles.size() == Indices.size() &&
21705 "Unmatched number of shufflevectors and indices");
21706
21707 auto *LI = dyn_cast<LoadInst>(Load);
21708 if (!LI)
21709 return false;
21710 assert(!Mask && GapMask.popcount() == Factor && "Unexpected mask on a load");
21711
21712 auto *VecTy = cast<FixedVectorType>(Shuffles[0]->getType());
21713 Type *EltTy = VecTy->getElementType();
21714
21715 const DataLayout &DL = LI->getDataLayout();
21716 Align Alignment = LI->getAlign();
21717
21718 // Skip if we do not have NEON and skip illegal vector types. We can
21719 // "legalize" wide vector types into multiple interleaved accesses as long as
21720 // the vector types are divisible by 128.
21721 if (!isLegalInterleavedAccessType(Factor, VecTy, Alignment, DL))
21722 return false;
21723
21724 unsigned NumLoads = getNumInterleavedAccesses(VecTy, DL);
21725
21726 // A pointer vector can not be the return type of the ldN intrinsics. Need to
21727 // load integer vectors first and then convert to pointer vectors.
21728 if (EltTy->isPointerTy())
21729 VecTy = FixedVectorType::get(DL.getIntPtrType(EltTy), VecTy);
21730
21731 IRBuilder<> Builder(LI);
21732
21733 // The base address of the load.
21734 Value *BaseAddr = LI->getPointerOperand();
21735
21736 if (NumLoads > 1) {
21737 // If we're going to generate more than one load, reset the sub-vector type
21738 // to something legal.
21739 VecTy = FixedVectorType::get(VecTy->getElementType(),
21740 VecTy->getNumElements() / NumLoads);
21741 }
21742
21743 assert(isTypeLegal(EVT::getEVT(VecTy)) && "Illegal vldN vector type!");
21744
21745 auto createLoadIntrinsic = [&](Value *BaseAddr) {
21746 if (Subtarget->hasNEON()) {
21747 Type *PtrTy = Builder.getPtrTy(LI->getPointerAddressSpace());
21748 Type *Tys[] = {VecTy, PtrTy};
21749 static const Intrinsic::ID LoadInts[3] = {Intrinsic::arm_neon_vld2,
21750 Intrinsic::arm_neon_vld3,
21751 Intrinsic::arm_neon_vld4};
21752
21754 Ops.push_back(BaseAddr);
21755 Ops.push_back(Builder.getInt32(LI->getAlign().value()));
21756
21757 return Builder.CreateIntrinsic(LoadInts[Factor - 2], Tys, Ops,
21758 /*FMFSource=*/nullptr, "vldN");
21759 } else {
21760 assert((Factor == 2 || Factor == 4) &&
21761 "expected interleave factor of 2 or 4 for MVE");
21762 Intrinsic::ID LoadInts =
21763 Factor == 2 ? Intrinsic::arm_mve_vld2q : Intrinsic::arm_mve_vld4q;
21764 Type *PtrTy = Builder.getPtrTy(LI->getPointerAddressSpace());
21765 Type *Tys[] = {VecTy, PtrTy};
21766
21768 Ops.push_back(BaseAddr);
21769 return Builder.CreateIntrinsic(LoadInts, Tys, Ops, /*FMFSource=*/nullptr,
21770 "vldN");
21771 }
21772 };
21773
21774 // Holds sub-vectors extracted from the load intrinsic return values. The
21775 // sub-vectors are associated with the shufflevector instructions they will
21776 // replace.
21778
21779 for (unsigned LoadCount = 0; LoadCount < NumLoads; ++LoadCount) {
21780 // If we're generating more than one load, compute the base address of
21781 // subsequent loads as an offset from the previous.
21782 if (LoadCount > 0)
21783 BaseAddr = Builder.CreateConstGEP1_32(VecTy->getElementType(), BaseAddr,
21784 VecTy->getNumElements() * Factor);
21785
21786 Value *VldN = createLoadIntrinsic(BaseAddr);
21787
21788 // Replace uses of each shufflevector with the corresponding vector loaded
21789 // by ldN.
21790 for (unsigned i = 0; i < Shuffles.size(); i++) {
21791 ShuffleVectorInst *SV = Shuffles[i];
21792 unsigned Index = Indices[i];
21793
21794 Value *SubVec = Builder.CreateExtractValue(VldN, Index);
21795
21796 // Convert the integer vector to pointer vector if the element is pointer.
21797 if (EltTy->isPointerTy())
21798 SubVec = Builder.CreateIntToPtr(
21799 SubVec,
21800 FixedVectorType::get(SV->getType()->getElementType(), VecTy));
21801
21802 SubVecs[SV].push_back(SubVec);
21803 }
21804 }
21805
21806 // Replace uses of the shufflevector instructions with the sub-vectors
21807 // returned by the load intrinsic. If a shufflevector instruction is
21808 // associated with more than one sub-vector, those sub-vectors will be
21809 // concatenated into a single wide vector.
21810 for (ShuffleVectorInst *SVI : Shuffles) {
21811 auto &SubVec = SubVecs[SVI];
21812 auto *WideVec =
21813 SubVec.size() > 1 ? concatenateVectors(Builder, SubVec) : SubVec[0];
21814 SVI->replaceAllUsesWith(WideVec);
21815 }
21816
21817 return true;
21818}
21819
21820/// Lower an interleaved store into a vstN intrinsic.
21821///
21822/// E.g. Lower an interleaved store (Factor = 3):
21823/// %i.vec = shuffle <8 x i32> %v0, <8 x i32> %v1,
21824/// <0, 4, 8, 1, 5, 9, 2, 6, 10, 3, 7, 11>
21825/// store <12 x i32> %i.vec, <12 x i32>* %ptr, align 4
21826///
21827/// Into:
21828/// %sub.v0 = shuffle <8 x i32> %v0, <8 x i32> v1, <0, 1, 2, 3>
21829/// %sub.v1 = shuffle <8 x i32> %v0, <8 x i32> v1, <4, 5, 6, 7>
21830/// %sub.v2 = shuffle <8 x i32> %v0, <8 x i32> v1, <8, 9, 10, 11>
21831/// call void llvm.arm.neon.vst3(%ptr, %sub.v0, %sub.v1, %sub.v2, 4)
21832///
21833/// Note that the new shufflevectors will be removed and we'll only generate one
21834/// vst3 instruction in CodeGen.
21835///
21836/// Example for a more general valid mask (Factor 3). Lower:
21837/// %i.vec = shuffle <32 x i32> %v0, <32 x i32> %v1,
21838/// <4, 32, 16, 5, 33, 17, 6, 34, 18, 7, 35, 19>
21839/// store <12 x i32> %i.vec, <12 x i32>* %ptr
21840///
21841/// Into:
21842/// %sub.v0 = shuffle <32 x i32> %v0, <32 x i32> v1, <4, 5, 6, 7>
21843/// %sub.v1 = shuffle <32 x i32> %v0, <32 x i32> v1, <32, 33, 34, 35>
21844/// %sub.v2 = shuffle <32 x i32> %v0, <32 x i32> v1, <16, 17, 18, 19>
21845/// call void llvm.arm.neon.vst3(%ptr, %sub.v0, %sub.v1, %sub.v2, 4)
21847 Value *LaneMask,
21848 ShuffleVectorInst *SVI,
21849 unsigned Factor,
21850 const APInt &GapMask) const {
21851 assert(Factor >= 2 && Factor <= getMaxSupportedInterleaveFactor() &&
21852 "Invalid interleave factor");
21853 auto *SI = dyn_cast<StoreInst>(Store);
21854 if (!SI)
21855 return false;
21856 assert(!LaneMask && GapMask.popcount() == Factor &&
21857 "Unexpected mask on store");
21858
21859 auto *VecTy = cast<FixedVectorType>(SVI->getType());
21860 assert(VecTy->getNumElements() % Factor == 0 && "Invalid interleaved store");
21861
21862 unsigned LaneLen = VecTy->getNumElements() / Factor;
21863 Type *EltTy = VecTy->getElementType();
21864 auto *SubVecTy = FixedVectorType::get(EltTy, LaneLen);
21865
21866 const DataLayout &DL = SI->getDataLayout();
21867 Align Alignment = SI->getAlign();
21868
21869 // Skip if we do not have NEON and skip illegal vector types. We can
21870 // "legalize" wide vector types into multiple interleaved accesses as long as
21871 // the vector types are divisible by 128.
21872 if (!isLegalInterleavedAccessType(Factor, SubVecTy, Alignment, DL))
21873 return false;
21874
21875 unsigned NumStores = getNumInterleavedAccesses(SubVecTy, DL);
21876
21877 Value *Op0 = SVI->getOperand(0);
21878 Value *Op1 = SVI->getOperand(1);
21879 IRBuilder<> Builder(SI);
21880
21881 // StN intrinsics don't support pointer vectors as arguments. Convert pointer
21882 // vectors to integer vectors.
21883 if (EltTy->isPointerTy()) {
21884 Type *IntTy = DL.getIntPtrType(EltTy);
21885
21886 // Convert to the corresponding integer vector.
21887 auto *IntVecTy =
21889 Op0 = Builder.CreatePtrToInt(Op0, IntVecTy);
21890 Op1 = Builder.CreatePtrToInt(Op1, IntVecTy);
21891
21892 SubVecTy = FixedVectorType::get(IntTy, LaneLen);
21893 }
21894
21895 // The base address of the store.
21896 Value *BaseAddr = SI->getPointerOperand();
21897
21898 if (NumStores > 1) {
21899 // If we're going to generate more than one store, reset the lane length
21900 // and sub-vector type to something legal.
21901 LaneLen /= NumStores;
21902 SubVecTy = FixedVectorType::get(SubVecTy->getElementType(), LaneLen);
21903 }
21904
21905 assert(isTypeLegal(EVT::getEVT(SubVecTy)) && "Illegal vstN vector type!");
21906
21907 auto Mask = SVI->getShuffleMask();
21908
21909 auto createStoreIntrinsic = [&](Value *BaseAddr,
21910 SmallVectorImpl<Value *> &Shuffles) {
21911 if (Subtarget->hasNEON()) {
21912 static const Intrinsic::ID StoreInts[3] = {Intrinsic::arm_neon_vst2,
21913 Intrinsic::arm_neon_vst3,
21914 Intrinsic::arm_neon_vst4};
21915 Type *PtrTy = Builder.getPtrTy(SI->getPointerAddressSpace());
21916 Type *Tys[] = {PtrTy, SubVecTy};
21917
21919 Ops.push_back(BaseAddr);
21920 append_range(Ops, Shuffles);
21921 Ops.push_back(Builder.getInt32(SI->getAlign().value()));
21922 Builder.CreateIntrinsic(StoreInts[Factor - 2], Tys, Ops);
21923 } else {
21924 assert((Factor == 2 || Factor == 4) &&
21925 "expected interleave factor of 2 or 4 for MVE");
21926 Intrinsic::ID StoreInts =
21927 Factor == 2 ? Intrinsic::arm_mve_vst2q : Intrinsic::arm_mve_vst4q;
21928 Type *PtrTy = Builder.getPtrTy(SI->getPointerAddressSpace());
21929 Type *Tys[] = {PtrTy, SubVecTy};
21930
21932 Ops.push_back(BaseAddr);
21933 append_range(Ops, Shuffles);
21934 for (unsigned F = 0; F < Factor; F++) {
21935 Ops.push_back(Builder.getInt32(F));
21936 Builder.CreateIntrinsic(StoreInts, Tys, Ops);
21937 Ops.pop_back();
21938 }
21939 }
21940 };
21941
21942 for (unsigned StoreCount = 0; StoreCount < NumStores; ++StoreCount) {
21943 // If we generating more than one store, we compute the base address of
21944 // subsequent stores as an offset from the previous.
21945 if (StoreCount > 0)
21946 BaseAddr = Builder.CreateConstGEP1_32(SubVecTy->getElementType(),
21947 BaseAddr, LaneLen * Factor);
21948
21949 SmallVector<Value *, 4> Shuffles;
21950
21951 // Split the shufflevector operands into sub vectors for the new vstN call.
21952 for (unsigned i = 0; i < Factor; i++) {
21953 unsigned IdxI = StoreCount * LaneLen * Factor + i;
21954 if (Mask[IdxI] >= 0) {
21955 Shuffles.push_back(Builder.CreateShuffleVector(
21956 Op0, Op1, createSequentialMask(Mask[IdxI], LaneLen, 0)));
21957 } else {
21958 unsigned StartMask = 0;
21959 for (unsigned j = 1; j < LaneLen; j++) {
21960 unsigned IdxJ = StoreCount * LaneLen * Factor + j;
21961 if (Mask[IdxJ * Factor + IdxI] >= 0) {
21962 StartMask = Mask[IdxJ * Factor + IdxI] - IdxJ;
21963 break;
21964 }
21965 }
21966 // Note: If all elements in a chunk are undefs, StartMask=0!
21967 // Note: Filling undef gaps with random elements is ok, since
21968 // those elements were being written anyway (with undefs).
21969 // In the case of all undefs we're defaulting to using elems from 0
21970 // Note: StartMask cannot be negative, it's checked in
21971 // isReInterleaveMask
21972 Shuffles.push_back(Builder.CreateShuffleVector(
21973 Op0, Op1, createSequentialMask(StartMask, LaneLen, 0)));
21974 }
21975 }
21976
21977 createStoreIntrinsic(BaseAddr, Shuffles);
21978 }
21979 return true;
21980}
21981
21989
21991 uint64_t &Members) {
21992 if (auto *ST = dyn_cast<StructType>(Ty)) {
21993 for (unsigned i = 0; i < ST->getNumElements(); ++i) {
21994 uint64_t SubMembers = 0;
21995 if (!isHomogeneousAggregate(ST->getElementType(i), Base, SubMembers))
21996 return false;
21997 Members += SubMembers;
21998 }
21999 } else if (auto *AT = dyn_cast<ArrayType>(Ty)) {
22000 uint64_t SubMembers = 0;
22001 if (!isHomogeneousAggregate(AT->getElementType(), Base, SubMembers))
22002 return false;
22003 Members += SubMembers * AT->getNumElements();
22004 } else if (Ty->isFloatTy()) {
22005 if (Base != HA_UNKNOWN && Base != HA_FLOAT)
22006 return false;
22007 Members = 1;
22008 Base = HA_FLOAT;
22009 } else if (Ty->isDoubleTy()) {
22010 if (Base != HA_UNKNOWN && Base != HA_DOUBLE)
22011 return false;
22012 Members = 1;
22013 Base = HA_DOUBLE;
22014 } else if (auto *VT = dyn_cast<VectorType>(Ty)) {
22015 Members = 1;
22016 switch (Base) {
22017 case HA_FLOAT:
22018 case HA_DOUBLE:
22019 return false;
22020 case HA_VECT64:
22021 return VT->getPrimitiveSizeInBits().getFixedValue() == 64;
22022 case HA_VECT128:
22023 return VT->getPrimitiveSizeInBits().getFixedValue() == 128;
22024 case HA_UNKNOWN:
22025 switch (VT->getPrimitiveSizeInBits().getFixedValue()) {
22026 case 64:
22027 Base = HA_VECT64;
22028 return true;
22029 case 128:
22030 Base = HA_VECT128;
22031 return true;
22032 default:
22033 return false;
22034 }
22035 }
22036 }
22037
22038 return (Members > 0 && Members <= 4);
22039}
22040
22041/// Return the correct alignment for the current calling convention.
22043 Type *ArgTy, const DataLayout &DL) const {
22044 const Align ABITypeAlign = DL.getABITypeAlign(ArgTy);
22045 if (!ArgTy->isVectorTy())
22046 return ABITypeAlign;
22047
22048 // Avoid over-aligning vector parameters. It would require realigning the
22049 // stack and waste space for no real benefit.
22050 MaybeAlign StackAlign = DL.getStackAlignment();
22051 assert(StackAlign && "data layout string is missing stack alignment");
22052 return std::min(ABITypeAlign, *StackAlign);
22053}
22054
22055/// Return true if a type is an AAPCS-VFP homogeneous aggregate or one of
22056/// [N x i32] or [N x i64]. This allows front-ends to skip emitting padding when
22057/// passing according to AAPCS rules.
22059 Type *Ty, CallingConv::ID CallConv, bool isVarArg,
22060 const DataLayout &DL) const {
22061 if (getEffectiveCallingConv(CallConv, isVarArg) !=
22063 return false;
22064
22066 uint64_t Members = 0;
22067 bool IsHA = isHomogeneousAggregate(Ty, Base, Members);
22068 LLVM_DEBUG(dbgs() << "isHA: " << IsHA << " "; Ty->dump());
22069
22070 bool IsIntArray = Ty->isArrayTy() && Ty->getArrayElementType()->isIntegerTy();
22071 return IsHA || IsIntArray;
22072}
22073
22075 ExceptionHandling EH, const Constant *PersonalityFn) const {
22076 // Platforms which do not use SjLj EH may return values in these registers
22077 // via the personality function.
22078 return EH == ExceptionHandling::SjLj ? Register() : ARM::R0;
22079}
22080
22082 ExceptionHandling EH, const Constant *PersonalityFn) const {
22083 // Platforms which do not use SjLj EH may return values in these registers
22084 // via the personality function.
22085 return EH == ExceptionHandling::SjLj ? Register() : ARM::R1;
22086}
22087
22088void ARMTargetLowering::initializeSplitCSR(MachineBasicBlock *Entry) const {
22089 // Update IsSplitCSR in ARMFunctionInfo.
22090 ARMFunctionInfo *AFI = Entry->getParent()->getInfo<ARMFunctionInfo>();
22091 AFI->setIsSplitCSR(true);
22092}
22093
22094void ARMTargetLowering::insertCopiesSplitCSR(
22095 MachineBasicBlock *Entry,
22096 const SmallVectorImpl<MachineBasicBlock *> &Exits) const {
22097 const ARMBaseRegisterInfo *TRI = Subtarget->getRegisterInfo();
22098 const MCPhysReg *IStart = TRI->getCalleeSavedRegsViaCopy(Entry->getParent());
22099 if (!IStart)
22100 return;
22101
22102 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
22103 MachineRegisterInfo *MRI = &Entry->getParent()->getRegInfo();
22104 MachineBasicBlock::iterator MBBI = Entry->begin();
22105 for (const MCPhysReg *I = IStart; *I; ++I) {
22106 const TargetRegisterClass *RC = nullptr;
22107 if (ARM::GPRRegClass.contains(*I))
22108 RC = &ARM::GPRRegClass;
22109 else if (ARM::DPRRegClass.contains(*I))
22110 RC = &ARM::DPRRegClass;
22111 else
22112 llvm_unreachable("Unexpected register class in CSRsViaCopy!");
22113
22114 Register NewVR = MRI->createVirtualRegister(RC);
22115 // Create copy from CSR to a virtual register.
22116 // FIXME: this currently does not emit CFI pseudo-instructions, it works
22117 // fine for CXX_FAST_TLS since the C++-style TLS access functions should be
22118 // nounwind. If we want to generalize this later, we may need to emit
22119 // CFI pseudo-instructions.
22120 assert(Entry->getParent()->getFunction().hasFnAttribute(
22121 Attribute::NoUnwind) &&
22122 "Function should be nounwind in insertCopiesSplitCSR!");
22123 Entry->addLiveIn(*I);
22124 BuildMI(*Entry, MBBI, DebugLoc(), TII->get(TargetOpcode::COPY), NewVR)
22125 .addReg(*I);
22126
22127 // Insert the copy-back instructions right before the terminator.
22128 for (auto *Exit : Exits)
22129 BuildMI(*Exit, Exit->getFirstTerminator(), DebugLoc(),
22130 TII->get(TargetOpcode::COPY), *I)
22131 .addReg(NewVR);
22132 }
22133}
22134
22139
22141 return Subtarget->hasMVEIntegerOps();
22142}
22143
22145 ComplexDeinterleavingOperation Operation, Type *Ty) const {
22146 auto *VTy = dyn_cast<FixedVectorType>(Ty);
22147 if (!VTy)
22148 return false;
22149
22150 auto *ScalarTy = VTy->getScalarType();
22151 unsigned NumElements = VTy->getNumElements();
22152
22153 unsigned VTyWidth = VTy->getScalarSizeInBits() * NumElements;
22154 if (VTyWidth < 128 || !llvm::isPowerOf2_32(VTyWidth))
22155 return false;
22156
22157 // Both VCADD and VCMUL/VCMLA support the same types, F16 and F32
22158 if (ScalarTy->isHalfTy() || ScalarTy->isFloatTy())
22159 return Subtarget->hasMVEFloatOps();
22160
22161 if (Operation != ComplexDeinterleavingOperation::CAdd)
22162 return false;
22163
22164 return Subtarget->hasMVEIntegerOps() &&
22165 (ScalarTy->isIntegerTy(8) || ScalarTy->isIntegerTy(16) ||
22166 ScalarTy->isIntegerTy(32));
22167}
22168
22170 static const MCPhysReg RCRegs[] = {ARM::FPSCR_RM};
22171 return RCRegs;
22172}
22173
22176 ComplexDeinterleavingRotation Rotation, Value *InputA, Value *InputB,
22177 Value *Accumulator) const {
22178
22180
22181 unsigned TyWidth = Ty->getScalarSizeInBits() * Ty->getNumElements();
22182
22183 assert(TyWidth >= 128 && "Width of vector type must be at least 128 bits");
22184
22185 if (TyWidth > 128) {
22186 int Stride = Ty->getNumElements() / 2;
22187 auto SplitSeq = llvm::seq<int>(0, Ty->getNumElements());
22188 auto SplitSeqVec = llvm::to_vector(SplitSeq);
22189 ArrayRef<int> LowerSplitMask(&SplitSeqVec[0], Stride);
22190 ArrayRef<int> UpperSplitMask(&SplitSeqVec[Stride], Stride);
22191
22192 auto *LowerSplitA = B.CreateShuffleVector(InputA, LowerSplitMask);
22193 auto *LowerSplitB = B.CreateShuffleVector(InputB, LowerSplitMask);
22194 auto *UpperSplitA = B.CreateShuffleVector(InputA, UpperSplitMask);
22195 auto *UpperSplitB = B.CreateShuffleVector(InputB, UpperSplitMask);
22196 Value *LowerSplitAcc = nullptr;
22197 Value *UpperSplitAcc = nullptr;
22198
22199 if (Accumulator) {
22200 LowerSplitAcc = B.CreateShuffleVector(Accumulator, LowerSplitMask);
22201 UpperSplitAcc = B.CreateShuffleVector(Accumulator, UpperSplitMask);
22202 }
22203
22204 auto *LowerSplitInt = createComplexDeinterleavingIR(
22205 B, OperationType, Rotation, LowerSplitA, LowerSplitB, LowerSplitAcc);
22206 auto *UpperSplitInt = createComplexDeinterleavingIR(
22207 B, OperationType, Rotation, UpperSplitA, UpperSplitB, UpperSplitAcc);
22208
22209 ArrayRef<int> JoinMask(&SplitSeqVec[0], Ty->getNumElements());
22210 return B.CreateShuffleVector(LowerSplitInt, UpperSplitInt, JoinMask);
22211 }
22212
22213 auto *IntTy = Type::getInt32Ty(B.getContext());
22214
22215 ConstantInt *ConstRotation = nullptr;
22216 if (OperationType == ComplexDeinterleavingOperation::CMulPartial) {
22217 ConstRotation = ConstantInt::get(IntTy, (int)Rotation);
22218
22219 if (Accumulator)
22220 return B.CreateIntrinsic(Intrinsic::arm_mve_vcmlaq, Ty,
22221 {ConstRotation, Accumulator, InputB, InputA});
22222 return B.CreateIntrinsic(Intrinsic::arm_mve_vcmulq, Ty,
22223 {ConstRotation, InputB, InputA});
22224 }
22225
22226 if (OperationType == ComplexDeinterleavingOperation::CAdd) {
22227 // 1 means the value is not halved.
22228 auto *ConstHalving = ConstantInt::get(IntTy, 1);
22229
22231 ConstRotation = ConstantInt::get(IntTy, 0);
22233 ConstRotation = ConstantInt::get(IntTy, 1);
22234
22235 if (!ConstRotation)
22236 return nullptr; // Invalid rotation for arm_mve_vcaddq
22237
22238 return B.CreateIntrinsic(Intrinsic::arm_mve_vcaddq, Ty,
22239 {ConstHalving, ConstRotation, InputA, InputB});
22240 }
22241
22242 return nullptr;
22243}
static bool isAddSubSExt(SDValue N, SelectionDAG &DAG)
static bool isVShiftRImm(SDValue Op, EVT VT, bool isNarrow, int64_t &Cnt)
isVShiftRImm - Check if this is a valid build_vector for the immediate operand of a vector shift righ...
static bool isExtendedBUILD_VECTOR(SDValue N, SelectionDAG &DAG, bool isSigned)
static SDValue carryFlagToValue(SDValue Glue, EVT VT, SelectionDAG &DAG, bool Invert)
static SDValue overflowFlagToValue(SDValue Glue, EVT VT, SelectionDAG &DAG)
static bool isZeroExtended(SDValue N, SelectionDAG &DAG)
static bool isCMN(SDValue Op, ISD::CondCode CC, SelectionDAG &DAG)
static const MCPhysReg GPRArgRegs[]
static SDValue valueToCarryFlag(SDValue Value, SelectionDAG &DAG, bool Invert)
constexpr MVT FlagsVT
Value type used for NZCV flags.
static unsigned getCmpOperandFoldingProfit(SDValue Op, bool AllowExtend)
Returns how profitable it is to fold a comparison's operand's shift and/or extension operations.
static bool getVShiftImm(SDValue Op, unsigned ElementBits, int64_t &Cnt)
getVShiftImm - Check if this is a valid build_vector for the immediate operand of a vector shift oper...
static bool optimizeLogicalImm(SDValue Op, unsigned Size, uint64_t Imm, const APInt &Demanded, TargetLowering::TargetLoweringOpt &TLO, unsigned NewOpc)
static bool isSafeSignedCMN(SDValue Op, SelectionDAG &DAG)
static SDValue LowerPREFETCH(SDValue Op, SelectionDAG &DAG)
static bool isSignExtended(SDValue N, SelectionDAG &DAG)
static bool isAddSubZExt(SDValue N, SelectionDAG &DAG)
static bool isVShiftLImm(SDValue Op, EVT VT, bool isLong, int64_t &Cnt)
isVShiftLImm - Check if this is a valid build_vector for the immediate operand of a vector shift left...
static bool canGuaranteeTCO(CallingConv::ID CC, bool GuaranteeTailCalls)
Return true if the calling convention is one that we can guarantee TCO for.
unsigned RegSize
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
amdgpu aa AMDGPU Address space based Alias Analysis Wrapper
unsigned Imm
unsigned uint64_t
static bool isConstant(const MachineInstr &MI)
constexpr LLT F64
constexpr LLT S1
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
static SDValue LowerVASTART(SDValue Op, SelectionDAG &DAG)
static bool isStore(int Opcode)
static bool isThumb(const MCSubtargetInfo &STI)
static SDValue PerformExtractEltToVMOVRRD(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool isIncompatibleReg(const MCPhysReg &PR, MVT VT)
static SDValue PerformVQDMULHCombine(SDNode *N, SelectionDAG &DAG)
static SDValue LowerBUILD_VECTOR_i1(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue LowerShift(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue LowerVECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue AddRequiredExtensionForVMULL(SDValue N, SelectionDAG &DAG, const EVT &OrigTy, const EVT &ExtTy, unsigned ExtOpcode)
AddRequiredExtensionForVMULL - Add a sign/zero extension to extend the total value size to 64 bits.
static cl::opt< unsigned > ConstpoolPromotionMaxSize("arm-promote-constant-max-size", cl::Hidden, cl::desc("Maximum size of constant to promote into a constant pool"), cl::init(64))
static bool isZeroOrAllOnes(SDValue N, bool AllOnes)
static SDValue LowerINSERT_VECTOR_ELT_i1(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static bool isVTBLMask(ArrayRef< int > M, EVT VT)
static SDValue PerformSUBCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformSUBCombine - Target-specific dag combine xforms for ISD::SUB.
static cl::opt< bool > EnableConstpoolPromotion("arm-promote-constant", cl::Hidden, cl::desc("Enable / disable promotion of unnamed_addr constants into " "constant pools"), cl::init(false))
static SDValue PerformFAddVSelectCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformExtractFpToIntStores(StoreSDNode *St, SelectionDAG &DAG)
static SDValue PerformVDUPCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
PerformVDUPCombine - Target-specific dag combine xforms for ARMISD::VDUP.
static SDValue PerformExtractEltCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *ST)
static const APInt * isPowerOf2Constant(SDValue V)
static SDValue PerformVCVTCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
PerformVCVTCombine - VCVT (floating-point to fixed-point, Advanced SIMD) can replace combinations of ...
static SDValue PerformVMOVhrCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue LowerVectorFP_TO_INT(SDValue Op, SelectionDAG &DAG)
static SDValue LowerVECTOR_SHUFFLEUsingOneOff(SDValue Op, ArrayRef< int > ShuffleMask, SelectionDAG &DAG)
static bool isValidMVECond(unsigned CC, bool IsFloat)
static SDValue PerformPREDICATE_CASTCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static ARMCC::CondCodes IntCCToARMCC(ISD::CondCode CC)
IntCCToARMCC - Convert a DAG integer condition code to an ARM CC.
static SDValue PerformSTORECombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformSTORECombine - Target-specific dag combine xforms for ISD::STORE.
static SDValue LowerCONCAT_VECTORS(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static bool isGTorGE(ISD::CondCode CC)
static bool CombineVLDDUP(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
CombineVLDDUP - For a VDUPLANE node N, check if its source operand is a vldN-lane (N > 1) intrinsic,...
static SDValue ParseBFI(SDNode *N, APInt &ToMask, APInt &FromMask)
static bool isReverseMask(ArrayRef< int > M, EVT VT)
static SDValue PerformSELECTCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue AddCombineTo64bitUMAAL(SDNode *AddeNode, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformVECTOR_REG_CASTCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformVMulVCTPCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
PerformVMulVCTPCombine - VCVT (fixed-point to floating-point, Advanced SIMD) can replace combinations...
static SDValue createGPRPairNode2xi32(SelectionDAG &DAG, SDValue V0, SDValue V1)
static SDValue bitcastf32Toi32(SDValue Op, SelectionDAG &DAG)
static bool findPointerConstIncrement(SDNode *N, SDValue *Ptr, SDValue *CInc)
static SDValue LowerEXTRACT_SUBVECTOR(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static bool CanInvertMVEVCMP(SDValue N)
static SDValue PerformLongShiftCombine(SDNode *N, SelectionDAG &DAG)
static SDValue AddCombineToVPADD(SDNode *N, SDValue N0, SDValue N1, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformShiftCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *ST)
PerformShiftCombine - Checks for immediate versions of vector shifts and lowers them.
static void FPCCToARMCC(ISD::CondCode CC, ARMCC::CondCodes &CondCode, ARMCC::CondCodes &CondCode2)
FPCCToARMCC - Convert a DAG fp condition code to an ARM CC.
static void ExpandREAD_REGISTER(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG)
static EVT getVectorTyFromPredicateVector(EVT VT)
static SDValue PerformFADDVCMLACombine(SDNode *N, SelectionDAG &DAG)
static SDValue handleCMSEValue(const SDValue &Value, const ISD::InputArg &Arg, SelectionDAG &DAG, const SDLoc &DL)
static SDValue PerformARMBUILD_VECTORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
Target-specific dag combine xforms for ARMISD::BUILD_VECTOR.
static bool isSRL16(const SDValue &Op)
static SDValue PerformVMOVrhCombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformLOADCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue IsCMPZCSINC(SDNode *Cmp, ARMCC::CondCodes &CC)
static unsigned getPointerConstIncrement(unsigned Opcode, SDValue Ptr, SDValue Inc, const SelectionDAG &DAG)
static SDValue combineSelectAndUseCommutative(SDNode *N, bool AllOnes, TargetLowering::DAGCombinerInfo &DCI)
static SDValue LowerATOMIC_FENCE(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static Register genTPEntry(MachineBasicBlock *TpEntry, MachineBasicBlock *TpLoopBody, MachineBasicBlock *TpExit, Register OpSizeReg, const TargetInstrInfo *TII, DebugLoc Dl, MachineRegisterInfo &MRI)
Adds logic in loop entry MBB to calculate loop iteration count and adds t2WhileLoopSetup and t2WhileL...
static SDValue createGPRPairNodei64(SelectionDAG &DAG, SDValue V)
static bool isLTorLE(ISD::CondCode CC)
static SDValue PerformVCMPCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformMVEVMULLCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue LowerSDIV_v4i16(SDValue N0, SDValue N1, const SDLoc &dl, SelectionDAG &DAG)
static SDValue performNegCMovCombine(SDNode *N, SelectionDAG &DAG)
static EVT getExtensionTo64Bits(const EVT &OrigVT)
static TargetLowering::ArgListTy getDivRemArgList(const SDNode *N, FunctionType *FuncTy, const AttributeList &FuncAttrs, RTLIB::LibcallImpl LCImpl)
static SDValue PerformBITCASTCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *ST)
static SDValue AddCombineTo64bitMLAL(SDNode *AddeSubeNode, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue LowerWRITE_REGISTER(SDValue Op, SelectionDAG &DAG)
static bool checkAndUpdateCPSRKill(MachineBasicBlock::iterator SelectItr, MachineBasicBlock *BB, const TargetRegisterInfo *TRI)
static SDValue PerformCMPZCombine(SDNode *N, SelectionDAG &DAG)
static bool hasNormalLoadOperand(SDNode *N)
hasNormalLoadOperand - Check if any of the operands of a BUILD_VECTOR node are normal,...
static SDValue PerformInsertEltCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
PerformInsertEltCombine - Target-specific dag combine xforms for ISD::INSERT_VECTOR_ELT.
static SDValue PerformVDUPLANECombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformVDUPLANECombine - Target-specific dag combine xforms for ARMISD::VDUPLANE.
static SDValue LowerBuildVectorOfFPTrunc(SDValue BV, SelectionDAG &DAG, const ARMSubtarget *ST)
static cl::opt< unsigned > ConstpoolPromotionMaxTotal("arm-promote-constant-max-total", cl::Hidden, cl::desc("Maximum size of ALL constants to promote into a constant pool"), cl::init(128))
static SDValue LowerTruncatei1(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static RTLIB::Libcall getDivRemLibcall(const SDNode *N, MVT::SimpleValueType SVT)
static SDValue SkipLoadExtensionForVMULL(LoadSDNode *LD, SelectionDAG &DAG)
SkipLoadExtensionForVMULL - return a load of the original vector size that does not do any sign/zero ...
static SDValue AddCombineVUZPToVPADDL(SDNode *N, SDValue N0, SDValue N1, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformADDCombineWithOperands(SDNode *N, SDValue N0, SDValue N1, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformADDCombineWithOperands - Try DAG combinations for an ADD with operands N0 and N1.
static SDValue PromoteMVEPredVector(SDLoc dl, SDValue Pred, EVT VT, SelectionDAG &DAG)
static SDValue matchCSET(unsigned &Opcode, bool &InvertCond, SDValue TrueVal, SDValue FalseVal, const ARMSubtarget *Subtarget)
static SDValue PerformORCombineToSMULWBT(SDNode *OR, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue LowerUDIV(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue FindBFIToCombineWith(SDNode *N)
static SDValue LowerADDSUBSAT(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static void checkVSELConstraints(ISD::CondCode CC, ARMCC::CondCodes &CondCode, bool &swpCmpOps, bool &swpVselOps)
static void ReplaceLongIntrinsic(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG)
static bool isS16(const SDValue &Op, SelectionDAG &DAG)
static bool isSRA16(const SDValue &Op)
static SDValue AddCombineBUILD_VECTORToVPADDL(SDNode *N, SDValue N0, SDValue N1, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue LowerVECTOR_SHUFFLEUsingMovs(SDValue Op, ArrayRef< int > ShuffleMask, SelectionDAG &DAG)
static SDValue LowerInterruptReturn(SmallVectorImpl< SDValue > &RetOps, const SDLoc &DL, SelectionDAG &DAG)
static SDValue LowerEXTRACT_VECTOR_ELT_i1(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue getInvertedARMCondCode(SDValue ARMcc, SelectionDAG &DAG)
static SDValue LowerSDIV_v4i8(SDValue X, SDValue Y, const SDLoc &dl, SelectionDAG &DAG)
static void expandf64Toi32(SDValue Op, SelectionDAG &DAG, SDValue &RetVal1, SDValue &RetVal2)
static SDValue LowerCONCAT_VECTORS_i1(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue LowerCTTZ(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformVLDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool isSHL16(const SDValue &Op)
static bool isVEXTMask(ArrayRef< int > M, EVT VT, bool &ReverseVEXT, unsigned &Imm)
static SDValue PerformMVEVLDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
cl::opt< unsigned > ArmMaxBaseUpdatesToCheck("arm-max-base-updates-to-check", cl::Hidden, cl::desc("Maximum number of base-updates to check generating postindex."), cl::init(64))
static bool isTruncMask(ArrayRef< int > M, EVT VT, bool Top, bool SingleSource)
static SDValue PerformADDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformADDCombine - Target-specific dag combine xforms for ISD::ADD.
static unsigned getLdOpcode(unsigned LdSize, bool IsThumb1, bool IsThumb2)
Return the load opcode for a given load size.
static SDValue LowerADDSUBO_CARRY(SDValue Op, SelectionDAG &DAG, unsigned Opcode, bool IsSigned)
static bool isLegalT2AddressImmediate(int64_t V, EVT VT, const ARMSubtarget *Subtarget)
static bool isLegalMVEShuffleOp(unsigned PFEntry)
static SDValue PerformSignExtendInregCombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformShuffleVMOVNCombine(ShuffleVectorSDNode *N, SelectionDAG &DAG)
static SDValue PerformVECTOR_SHUFFLECombine(SDNode *N, SelectionDAG &DAG)
PerformVECTOR_SHUFFLECombine - Target-specific dag combine xforms for ISD::VECTOR_SHUFFLE.
static SDValue SkipExtensionForVMULL(SDNode *N, SelectionDAG &DAG)
SkipExtensionForVMULL - For a node that is a SIGN_EXTEND, ZERO_EXTEND, ANY_EXTEND,...
static int getNegationCost(SDValue Op)
static bool isVMOVNTruncMask(ArrayRef< int > M, EVT ToVT, bool rev)
static SDValue PerformVQMOVNCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static MachineBasicBlock * OtherSucc(MachineBasicBlock *MBB, MachineBasicBlock *Succ)
static SDValue LowerVecReduceMinMax(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformFPExtendCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformAddcSubcCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformVSELECTCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformVECREDUCE_ADDCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue getZeroVector(EVT VT, SelectionDAG &DAG, const SDLoc &dl)
getZeroVector - Returns a vector of specified type with all zero elements.
static SDValue LowerAtomicLoadStore(SDValue Op, SelectionDAG &DAG)
static SDValue PerformSplittingToNarrowingStores(StoreSDNode *St, SelectionDAG &DAG)
static bool getT2IndexedAddressParts(SDNode *Ptr, EVT VT, bool isSEXTLoad, SDValue &Base, SDValue &Offset, bool &isInc, SelectionDAG &DAG)
static ARMCC::CondCodes getVCMPCondCode(SDValue N)
static cl::opt< bool > ARMInterworking("arm-interworking", cl::Hidden, cl::desc("Enable / disable ARM interworking (for debugging only)"), cl::init(true))
static void ReplaceREADCYCLECOUNTER(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformORCombineToBFI(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static bool isConditionalZeroOrAllOnes(SDNode *N, bool AllOnes, SDValue &CC, bool &Invert, SDValue &OtherOp, SelectionDAG &DAG)
static SDValue LowerEXTRACT_VECTOR_ELT(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformVSetCCToVCTPCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue LowerBUILD_VECTORToVIDUP(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static bool isZeroVector(SDValue N)
static SDValue PerformAddeSubeCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static void ReplaceCMP_SWAP_64Results(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG)
static bool isLowerSaturate(const SDValue LHS, const SDValue RHS, const SDValue TrueVal, const SDValue FalseVal, const ISD::CondCode CC, const SDValue K)
static bool isLegalLogicalImmediate(unsigned Imm, const ARMSubtarget *Subtarget)
static SDValue LowerPredicateLoad(SDValue Op, SelectionDAG &DAG)
static void emitPostSt(MachineBasicBlock *BB, MachineBasicBlock::iterator Pos, const TargetInstrInfo *TII, const DebugLoc &dl, unsigned StSize, unsigned Data, unsigned AddrIn, unsigned AddrOut, bool IsThumb1, bool IsThumb2)
Emit a post-increment store operation with given size.
static bool isVMOVNMask(ArrayRef< int > M, EVT VT, bool Top, bool SingleSource)
static SDValue CombineBaseUpdate(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
CombineBaseUpdate - Target-specific DAG combine function for VLDDUP, NEON load/store intrinsics,...
static SDValue LowerSaturatingConditional(SDValue Op, SelectionDAG &DAG)
static SDValue PerformSubCSINCCombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformVMOVRRDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformVMOVRRDCombine - Target-specific dag combine xforms for ARMISD::VMOVRRD.
static SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformCSETCombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformVMOVNCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue PerformInsertSubvectorCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static SDValue LowerVectorExtend(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue WinDBZCheckDenominator(SelectionDAG &DAG, SDNode *N, SDValue InChain)
static SDValue LowerVECTOR_SHUFFLEv8i8(SDValue Op, ArrayRef< int > ShuffleMask, SelectionDAG &DAG)
static SDValue PerformVMULCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformVMULCombine Distribute (A + B) * C to (A * C) + (B * C) to take advantage of the special multi...
static SDValue LowerMUL(SDValue Op, SelectionDAG &DAG)
static SDValue GeneratePerfectShuffle(unsigned PFEntry, SDValue LHS, SDValue RHS, SelectionDAG &DAG, const SDLoc &dl)
GeneratePerfectShuffle - Given an entry in the perfect-shuffle table, emit the specified operations t...
static SDValue PerformBFICombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformORCombine - Target-specific dag combine xforms for ISD::OR.
static SDValue LowerMLOAD(SDValue Op, SelectionDAG &DAG)
static SDValue PerformTruncatingStoreCombine(StoreSDNode *St, SelectionDAG &DAG)
static void emitPostLd(MachineBasicBlock *BB, MachineBasicBlock::iterator Pos, const TargetInstrInfo *TII, const DebugLoc &dl, unsigned LdSize, unsigned Data, unsigned AddrIn, unsigned AddrOut, bool IsThumb1, bool IsThumb2)
Emit a post-increment load operation with given size.
static SDValue TryDistrubutionADDVecReduce(SDNode *N, SelectionDAG &DAG)
static bool isValidBaseUpdate(SDNode *N, SDNode *User)
static SDValue IsSingleInstrConstant(SDValue N, SelectionDAG &DAG, const ARMSubtarget *ST, const SDLoc &dl)
static bool IsQRMVEInstruction(const SDNode *N, const SDNode *Op)
static SDValue PerformMinMaxToSatCombine(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformXORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static bool getMVEIndexedAddressParts(SDNode *Ptr, EVT VT, Align Alignment, bool isSEXTLoad, bool IsMasked, bool isLE, SDValue &Base, SDValue &Offset, bool &isInc, SelectionDAG &DAG)
std::pair< unsigned, const TargetRegisterClass * > RCPair
static SDValue combineSelectAndUse(SDNode *N, SDValue Slct, SDValue OtherOp, TargetLowering::DAGCombinerInfo &DCI, bool AllOnes=false)
static SDValue PerformExtendCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
PerformExtendCombine - Target-specific DAG combining for ISD::SIGN_EXTEND, ISD::ZERO_EXTEND,...
static SDValue LowerSDIV(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
cl::opt< unsigned > MVEMaxSupportedInterleaveFactor("mve-max-interleave-factor", cl::Hidden, cl::desc("Maximum interleave factor for MVE VLDn to generate."), cl::init(2))
static SDValue isVMOVModifiedImm(uint64_t SplatBits, uint64_t SplatUndef, unsigned SplatBitSize, SelectionDAG &DAG, const SDLoc &dl, EVT &VT, EVT VectorVT, VMOVModImmType type)
isVMOVModifiedImm - Check if the specified splat value corresponds to a valid vector constant for a N...
static SDValue LowerBuildVectorOfFPExt(SDValue BV, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue CombineVMOVDRRCandidateWithVecOp(const SDNode *BC, SelectionDAG &DAG)
BC is a bitcast that is about to be turned into a VMOVDRR.
static SDValue promoteToConstantPool(const ARMTargetLowering *TLI, const GlobalValue *GV, SelectionDAG &DAG, EVT PtrVT, const SDLoc &dl)
static unsigned isNEONTwoResultShuffleMask(ArrayRef< int > ShuffleMask, EVT VT, unsigned &WhichResult, bool &isV_UNDEF)
Check if ShuffleMask is a NEON two-result shuffle (VZIP, VUZP, VTRN), and return the corresponding AR...
static bool BitsProperlyConcatenate(const APInt &A, const APInt &B)
static bool getARMIndexedAddressParts(SDNode *Ptr, EVT VT, bool isSEXTLoad, SDValue &Base, SDValue &Offset, bool &isInc, SelectionDAG &DAG)
static SDValue LowerVecReduce(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue LowerVectorINT_TO_FP(SDValue Op, SelectionDAG &DAG)
static bool TryCombineBaseUpdate(struct BaseUpdateTarget &Target, struct BaseUpdateUser &User, bool SimpleConstIncOnly, TargetLowering::DAGCombinerInfo &DCI)
static bool allUsersAreInFunction(const Value *V, const Function *F)
Return true if all users of V are within function F, looking through ConstantExprs.
static bool isSingletonVEXTMask(ArrayRef< int > M, EVT VT, unsigned &Imm)
static SDValue PerformVMOVDRRCombine(SDNode *N, SelectionDAG &DAG)
PerformVMOVDRRCombine - Target-specific dag combine xforms for ARMISD::VMOVDRR.
static bool isLowerSaturatingConditional(const SDValue &Op, SDValue &V, SDValue &SatK)
static bool isLegalAddressImmediate(int64_t V, EVT VT, const ARMSubtarget *Subtarget)
isLegalAddressImmediate - Return true if the integer value can be used as the offset of the target ad...
static SDValue LowerVSETCC(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static bool isLegalT1AddressImmediate(int64_t V, EVT VT)
static SDValue CombineANDShift(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue LowerSETCCCARRY(SDValue Op, SelectionDAG &DAG)
static SDValue PerformSHLSimplify(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *ST)
static SDValue PerformADDECombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformADDECombine - Target-specific dag combine transform from ARMISD::ADDC, ARMISD::ADDE,...
static SDValue PerformReduceShuffleCombine(SDNode *N, SelectionDAG &DAG)
static SDValue PerformUMLALCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue LowerTruncate(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformHWLoopCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *ST)
static SDValue PerformORCombineToShiftInsert(SelectionDAG &DAG, SDValue AndOp, SDValue ShiftOp, EVT VT, SDLoc dl)
static SDValue PerformSplittingMVETruncToNarrowingStores(StoreSDNode *St, SelectionDAG &DAG)
static bool isHomogeneousAggregate(Type *Ty, HABaseType &Base, uint64_t &Members)
static SDValue PerformMULCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformFADDCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue LowerReverse_VECTOR_SHUFFLE(SDValue Op, SelectionDAG &DAG)
static SDValue PerformANDCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static SDValue PerformADDVecReduce(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue LowerPredicateStore(SDValue Op, SelectionDAG &DAG)
static SDValue SearchLoopIntrinsic(SDValue N, ISD::CondCode &CC, int &Imm, bool &Negate)
static bool canChangeToInt(SDValue Op, bool &SeenZero, const ARMSubtarget *Subtarget)
canChangeToInt - Given the fp compare operand, return true if it is suitable to morph to an integer c...
static unsigned getStOpcode(unsigned StSize, bool IsThumb1, bool IsThumb2)
Return the store opcode for a given store size.
static bool IsVUZPShuffleNode(SDNode *N)
static SDValue Expand64BitShift(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue AddCombineTo64BitSMLAL16(SDNode *AddcNode, SDNode *AddeNode, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
static void attachMEMCPYScratchRegs(const ARMSubtarget *Subtarget, MachineInstr &MI, const SDNode *Node)
Attaches vregs to MEMCPY that it will use as scratch registers when it is expanded into LDM/STM.
static bool isFloatingPointZero(SDValue Op)
isFloatingPointZero - Return true if this is +0.0.
static SDValue findMUL_LOHI(SDValue V)
static SDValue LowerVECTOR_SHUFFLE_i1(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformORCombine_i1(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *Subtarget)
static SDValue PerformSplittingMVEEXTToWideningLoad(SDNode *N, SelectionDAG &DAG)
static SDValue PerformSplittingToWideningLoad(SDNode *N, SelectionDAG &DAG)
static void genTPLoopBody(MachineBasicBlock *TpLoopBody, MachineBasicBlock *TpEntry, MachineBasicBlock *TpExit, const TargetInstrInfo *TII, DebugLoc Dl, MachineRegisterInfo &MRI, Register OpSrcReg, Register OpDestReg, Register ElementCountReg, Register TotalIterationsReg, bool IsMemcpy)
Adds logic in the loopBody MBB to generate MVE_VCTP, t2DoLoopDec and t2DoLoopEnd.
static SDValue PerformBUILD_VECTORCombine(SDNode *N, TargetLowering::DAGCombinerInfo &DCI, const ARMSubtarget *Subtarget)
PerformBUILD_VECTORCombine - Target-specific dag combine xforms for ISD::BUILD_VECTOR.
static SDValue LowerVecReduceF(SDValue Op, SelectionDAG &DAG, const ARMSubtarget *ST)
static SDValue PerformMinMaxCombine(SDNode *N, SelectionDAG &DAG, const ARMSubtarget *ST)
PerformMinMaxCombine - Target-specific DAG combining for creating truncating saturates.
arm ldst static false bool definesCPSR(const MachineInstr &MI)
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
This file a TargetTransformInfoImplBase conforming object specific to the ARM target machine.
Function Alias Analysis false
Function Alias Analysis Results
Atomic ordering constants.
This file contains the simple types necessary to represent the attributes associated with functions a...
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
This file implements the BitVector class.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static std::optional< bool > isBigEndian(const SmallDenseMap< int64_t, int64_t, 8 > &MemOffset2Idx, int64_t LowestIdx)
Given a map from byte offsets in memory to indices in a load/store, determine if that map corresponds...
This file contains the declarations for the subclasses of Constant, which represent the different fla...
static void createLoadIntrinsic(IntrinsicInst *II, LoadInst *LI, dxil::ResourceTypeInfo &RTI)
static void createStoreIntrinsic(IntrinsicInst *II, StoreInst *SI, dxil::ResourceTypeInfo &RTI)
This file defines the DenseMap class.
static bool isSigned(unsigned Opcode)
#define Check(C,...)
#define op(i)
#define im(i)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
std::pair< Value *, Value * > ShuffleOps
We are building a shuffle to create V, which is a sequence of insertelement, extractelement pairs.
static Value * LowerCTPOP(LLVMContext &Context, Value *V, Instruction *IP)
Emit the code to lower ctpop of V before the specified instruction IP.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
static LVOptions Options
Definition LVOptions.cpp:25
lazy value info
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
nvptx lower args
uint64_t High
uint64_t IntrinsicInst * II
R600 Clause Merge
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
SI Lower i1 Copies
Func MI getDebugLoc()))
This file contains some templates that are useful if you are working with the STL at all.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
static cl::opt< unsigned > MaxSteps("has-predecessor-max-steps", cl::Hidden, cl::init(8192), cl::desc("DAG combiner limit number of steps when searching DAG " "for predecessor nodes"))
This file defines the SmallPtrSet class.
This file defines the SmallVector class.
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
This file contains some functions that are useful when dealing with strings.
This file implements the StringSwitch template, which mimics a switch() statement whose cases are str...
#define LLVM_DEBUG(...)
Definition Debug.h:119
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
static SymbolRef::Type getType(const Symbol *Sym)
Definition TapiFile.cpp:39
This file describes how to lower LLVM code to machine code.
static X86::CondCode getSwappedCondition(X86::CondCode CC)
Assuming the flags are set by MI(a,b), return the condition code if we modify the instructions such t...
static constexpr int Concat[]
Value * RHS
Value * LHS
BinaryOperator * Mul
static bool isIntrinsic(const CallBase &Call, Intrinsic::ID ID)
The Input class is used to parse a yaml document into in-memory structs and vectors.
static constexpr roundingMode rmTowardZero
Definition APFloat.h:365
LLVM_ABI bool getExactInverse(APFloat *Inv) const
If this value is normal and has an exact, normal, multiplicative inverse, store it in inv and return ...
Definition APFloat.cpp:5976
APInt bitcastToAPInt() const
Definition APFloat.h:1475
opStatus convertToInteger(MutableArrayRef< integerPart > Input, unsigned int Width, bool IsSigned, roundingMode RM, bool *IsExact) const
Definition APFloat.h:1436
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
bool isMinSignedValue() const
Determine if this is the smallest signed value.
Definition APInt.h:419
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1532
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
bool sgt(const APInt &RHS) const
Signed greater than comparison.
Definition APInt.h:1205
bool isAllOnes() const
Determine if all bits are set. This is true for zero-width values.
Definition APInt.h:367
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool ult(const APInt &RHS) const
Unsigned less than comparison.
Definition APInt.h:1115
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
unsigned logBase2() const
Definition APInt.h:1781
uint64_t getLimitedValue(uint64_t Limit=UINT64_MAX) const
If this value is smaller than the specified limit, return it, otherwise return the limit value.
Definition APInt.h:471
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1261
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:436
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:302
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
bool isOne() const
Determine if this is a value of 1.
Definition APInt.h:385
static APInt getOneBitSet(unsigned numBits, unsigned BitNo)
Return an APInt with exactly one bit set in the result.
Definition APInt.h:235
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
void lshrInPlace(unsigned ShiftAmt)
Logical right-shift this APInt by ShiftAmt in place.
Definition APInt.h:860
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
unsigned countr_one() const
Count the number of trailing one bits.
Definition APInt.h:1676
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1225
An arbitrary precision integer that knows its signedness.
Definition APSInt.h:24
const ARMBaseRegisterInfo & getRegisterInfo() const
const uint32_t * getSjLjDispatchPreservedMask(const MachineFunction &MF) const
const MCPhysReg * getCalleeSavedRegs(const MachineFunction *MF) const override
Code Generation virtual methods...
Register getFrameRegister(const MachineFunction &MF) const override
const uint32_t * getCallPreservedMask(const MachineFunction &MF, CallingConv::ID) const override
const uint32_t * getTLSCallPreservedMask(const MachineFunction &MF) const
const uint32_t * getThisReturnPreservedMask(const MachineFunction &MF, CallingConv::ID) const
getThisReturnPreservedMask - Returns a call preserved mask specific to the case that 'returned' is on...
static ARMConstantPoolConstant * Create(const Constant *C, unsigned ID)
static ARMConstantPoolMBB * Create(LLVMContext &C, const MachineBasicBlock *mbb, unsigned ID, unsigned char PCAdj)
static ARMConstantPoolSymbol * Create(LLVMContext &C, StringRef s, unsigned ID, unsigned char PCAdj, ARMCP::ARMCPModifier Modifier=ARMCP::no_modifier, bool AddCurrentAddress=false)
ARMConstantPoolValue - ARM specific constantpool value.
ARMFunctionInfo - This class is derived from MachineFunctionInfo and contains private ARM-specific in...
SmallPtrSet< const GlobalVariable *, 2 > & getGlobalsPromotedToConstantPool()
void setArgumentStackToRestore(unsigned v)
void setArgRegsSaveSize(unsigned s)
void setReturnRegsCount(unsigned s)
unsigned getArgRegsSaveSize() const
void markGlobalAsPromotedToConstantPool(const GlobalVariable *GV)
Indicate to the backend that GV has had its storage changed to inside a constant pool.
void setArgumentStackSize(unsigned size)
unsigned getArgumentStackSize() const
const Triple & getTargetTriple() const
const ARMBaseInstrInfo * getInstrInfo() const override
bool isThumb1Only() const
bool useFPVFMx() const
bool isThumb2() const
bool hasBaseDSP() const
const ARMTargetLowering * getTargetLowering() const override
const ARMBaseRegisterInfo * getRegisterInfo() const override
bool hasVFP2Base() const
bool useFPVFMx64() const
bool isLittle() const
bool useFPVFMx16() const
bool isMClass() const
bool useMulOps() const
bool shouldFoldSelectWithIdentityConstant(unsigned BinOpcode, EVT VT, unsigned SelectOpcode, SDValue X, SDValue Y) const override
Return true if pulling a binary operation into a select with an identity constant is profitable.
bool isReadOnly(const GlobalValue *GV) const
unsigned getMaxSupportedInterleaveFactor() const override
Get the maximum supported factor for interleaved memory accesses.
TargetLoweringBase::AtomicExpansionKind shouldExpandAtomicLoadInIR(LoadInst *LI) const override
Returns how the given (atomic) load should be expanded by the IR-level AtomicExpand pass.
unsigned getNumInterleavedAccesses(VectorType *VecTy, const DataLayout &DL) const
Returns the number of interleaved accesses that will be generated when lowering accesses of the given...
bool shouldInsertFencesForAtomic(const Instruction *I) const override
Whether AtomicExpandPass should automatically insert fences and reduce ordering for this atomic.
Align getABIAlignmentForCallingConv(Type *ArgTy, const DataLayout &DL) const override
Return the correct alignment for the current calling convention.
bool isDesirableToCommuteWithShift(const SDNode *N, CombineLevel Level) const override
Return true if it is profitable to move this shift by a constant amount through its operand,...
ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const override
Examine constraint string and operand type and determine a weight value.
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
isLegalAddressingMode - Return true if the addressing mode represented by AM is legal for this target...
const ARMSubtarget * getSubtarget() const
bool isLegalT2ScaledAddressingMode(const AddrMode &AM, EVT VT) const
bool isLegalT1ScaledAddressingMode(const AddrMode &AM, EVT VT) const
Returns true if the addressing mode representing by AM is legal for the Thumb1 target,...
bool getPreIndexedAddressParts(SDNode *N, SDValue &Base, SDValue &Offset, ISD::MemIndexedMode &AM, SelectionDAG &DAG) const override
getPreIndexedAddressParts - returns true by value, base pointer and offset pointer and addressing mod...
MachineInstr * EmitKCFICheck(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator &MBBI, const TargetInstrInfo *TII) const override
bool shouldAlignPointerArgs(CallInst *CI, unsigned &MinSize, Align &PrefAlign) const override
Return true if the pointer arguments to CI should be aligned by aligning the object whose address is ...
void getTgtMemIntrinsic(SmallVectorImpl< IntrinsicInfo > &Infos, const CallBase &I, MachineFunction &MF, unsigned Intrinsic) const override
getTgtMemIntrinsic - Represent NEON load and store intrinsics as MemIntrinsicNodes.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
ReplaceNodeResults - Replace the results of node with an illegal result type with new values built ou...
void emitAtomicCmpXchgNoStoreLLBalance(IRBuilderBase &Builder) const override
bool isMulAddWithConstProfitable(SDValue AddNode, SDValue ConstNode) const override
Return true if it may be profitable to transform (mul (add x, c1), c2) -> (add (mul x,...
bool isLegalAddImmediate(int64_t Imm) const override
isLegalAddImmediate - Return true if the specified immediate is legal add immediate,...
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
Returns the target specific optimal type for load and store operations as a result of memset,...
Instruction * emitTrailingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const override
bool isFNegFree(EVT VT) const override
Return true if an fneg operation is free to the point where it is never worthwhile to replace it with...
void finalizeLowering(MachineFunction &MF) const override
Execute target specific actions to finalize target lowering.
SDValue PerformMVETruncCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize=false) const override
isFPImmLegal - Returns true if the target can instruction select the specified FP immediate natively.
ConstraintType getConstraintType(StringRef Constraint) const override
getConstraintType - Given a constraint letter, return the type of constraint it is for this target.
bool preferIncOfAddToSubOfNot(EVT VT) const override
These two forms are equivalent: sub y, (xor x, -1) add (add x, 1), y The variant with two add's is IR...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
TargetLoweringBase::AtomicExpansionKind shouldExpandAtomicStoreInIR(StoreInst *SI) const override
Returns how the given (atomic) store should be expanded by the IR-level AtomicExpand pass into.
SDValue PerformIntrinsicCombine(SDNode *N, DAGCombinerInfo &DCI) const
PerformIntrinsicCombine - ARM-specific DAG combining for intrinsics.
bool shouldFoldConstantShiftPairToMask(const SDNode *N) const override
Return true if it is profitable to fold a pair of shifts into a mask.
bool isDesirableToCommuteXorWithShift(const SDNode *N) const override
Return true if it is profitable to combine an XOR of a logical shift to create a logical shift of NOT...
SDValue PerformCMOVCombine(SDNode *N, SelectionDAG &DAG) const
PerformCMOVCombine - Target-specific DAG combining for ARMISD::CMOV.
TargetLoweringBase::AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *AI) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
Value * createComplexDeinterleavingIR(IRBuilderBase &B, ComplexDeinterleavingOperation OperationType, ComplexDeinterleavingRotation Rotation, Value *InputA, Value *InputB, Value *Accumulator=nullptr) const override
Create the IR node for the given complex deinterleaving operation.
bool isComplexDeinterleavingSupported() const override
Does this target support complex deinterleaving.
SDValue PerformMVEExtCombine(SDNode *N, DAGCombinerInfo &DCI) const
FastISel * createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo, const LibcallLoweringInfo *libcallLowering) const override
createFastISel - This method returns a target specific FastISel object, or null if the target does no...
void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const override
Inserts necessary declarations for SSP (stack protection) purpose.
bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &OriginalDemandedBits, const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth) const override
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
getSetCCResultType - Return the value type to use for ISD::SETCC.
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
Value * emitStoreConditional(IRBuilderBase &Builder, Value *Val, Value *Addr, AtomicOrdering Ord) const override
Perform a store-conditional operation to Addr.
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool isVarArg) const
void AdjustInstrPostInstrSelection(MachineInstr &MI, SDNode *Node) const override
This method should be implemented by targets that mark instructions with the 'hasPostISelHook' flag.
bool isTruncateFree(Type *SrcTy, Type *DstTy) const override
Return true if it's free to truncate a value of type FromTy to type ToTy.
bool isShuffleMaskLegal(ArrayRef< int > M, EVT VT) const override
isShuffleMaskLegal - Targets can use this to indicate that they only support some VECTOR_SHUFFLE oper...
Register getExceptionPointerRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception address on entry to an ...
TargetLoweringBase::AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *AI) const override
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
bool shouldConvertConstantLoadToIntImm(const APInt &Imm, Type *Ty) const override
Returns true if it is beneficial to convert a load of a constant to just the constant itself.
bool lowerInterleavedStore(Instruction *Store, Value *Mask, ShuffleVectorInst *SVI, unsigned Factor, const APInt &GapMask) const override
Lower an interleaved store into a vstN intrinsic.
const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const override
getRegClassFor - Return the register class that should be used for the specified value type.
bool useLoadStackGuardNode(const Module &M) const override
If this function returns true, SelectionDAGBuilder emits a LOAD_STACK_GUARD node when it is lowering ...
bool lowerInterleavedLoad(Instruction *Load, Value *Mask, ArrayRef< ShuffleVectorInst * > Shuffles, ArrayRef< unsigned > Indices, unsigned Factor, const APInt &GapMask) const override
Lower an interleaved load into a vldN intrinsic.
std::pair< const TargetRegisterClass *, uint8_t > findRepresentativeClass(const TargetRegisterInfo *TRI, MVT VT) const override
Return the largest legal super-reg register class of the register class for the specified type and it...
bool preferSelectsOverBooleanArithmetic(EVT VT) const override
Should we prefer selects to doing arithmetic on boolean types.
bool isZExtFree(SDValue Val, EVT VT2) const override
Return true if zero-extending the specific node Val to type VT2 is free (either because it's implicit...
bool isCheapToSpeculateCttz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic cttz.
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
bool isCheapToSpeculateCtlz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic ctlz.
bool targetShrinkDemandedConstant(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, TargetLoweringOpt &TLO) const override
bool hasAndNot(SDValue Y) const override
Return true if the target has a bitwise and-not operation: X = ~A & B This can be used to simplify se...
Register getExceptionSelectorRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception typeid on entry to a la...
ExtractSubvectorCost getExtractSubvectorCost(EVT ResVT, EVT SrcVT, unsigned Index) const override
Return the cost of EXTRACT_SUBVECTOR for this result type with this index.
CallingConv::ID getEffectiveCallingConv(CallingConv::ID CC, bool isVarArg) const
getEffectiveCallingConv - Get the effective calling convention, taking into account presence of float...
ARMTargetLowering(const TargetMachine &TM, const ARMSubtarget &STI)
bool isComplexDeinterleavingOperationSupported(ComplexDeinterleavingOperation Operation, Type *Ty) const override
Does this target support complex deinterleaving with the given operation and type.
bool supportKCFIBundles() const override
Return true if the target supports kcfi operand bundles.
SDValue PerformBRCONDCombine(SDNode *N, SelectionDAG &DAG) const
PerformBRCONDCombine - Target-specific DAG combining for ARMISD::BRCOND.
Type * shouldConvertSplatType(ShuffleVectorInst *SVI) const override
Given a shuffle vector SVI representing a vector splat, return a new scalar type of size equal to SVI...
Value * emitLoadLinked(IRBuilderBase &Builder, Type *ValueTy, Value *Addr, AtomicOrdering Ord) const override
Perform a load-linked operation on Addr, returning a "Value *" with the corresponding pointee type.
Instruction * makeDMB(IRBuilderBase &Builder, ARM_MB::MemBOpt Domain) const
bool isLegalICmpImmediate(int64_t Imm) const override
isLegalICmpImmediate - Return true if the specified immediate is legal icmp immediate,...
const char * LowerXConstraint(EVT ConstraintVT) const override
Try to replace an X constraint, which matches anything, with another that has more specific requireme...
unsigned getJumpTableEncoding() const override
Return the entry encoding for a jump table in the current function.
bool isDesirableToTransformToIntegerOp(unsigned Opc, EVT VT) const override
Return true if it is profitable for dag combiner to transform a floating point op of specified opcode...
CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool isVarArg) const
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags, unsigned *Fast) const override
allowsMisalignedMemoryAccesses - Returns true if the target allows unaligned memory accesses of the s...
bool isLegalInterleavedAccessType(unsigned Factor, FixedVectorType *VecTy, Align Alignment, const DataLayout &DL) const
Returns true if VecTy is a legal interleaved access type.
bool isVectorLoadExtDesirable(SDValue ExtVal) const override
Return true if folding a vector load into ExtVal (a sign, zero, or any extend node) is profitable.
bool canCombineStoreAndExtract(Type *VectorTy, Value *Idx, unsigned &Cost) const override
Return true if the target can combine store(extractelement VectorTy,Idx).
bool useSoftFloat() const override
bool alignLoopsWithOptSize() const override
Should loops be aligned even when the function is marked OptSize (but not MinSize).
SDValue PerformCMOVToBFICombine(SDNode *N, SelectionDAG &DAG) const
bool allowTruncateForTailCall(Type *Ty1, Type *Ty2) const override
Return true if a truncation from FromTy to ToTy is permitted when deciding whether a call is in tail ...
void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const override
LowerAsmOperandForConstraint - Lower the specified operand into the Ops vector.
bool hasAndNotCompare(SDValue V) const override
Return true if the target should transform: (X & Y) == Y ---> (~X & Y) == 0 (X & Y) !...
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
bool shouldConvertFpToSat(unsigned Op, EVT FPVT, EVT VT) const override
Should we generate fp_to_si_sat and fp_to_ui_sat from type FPVT to type VT.
bool functionArgumentNeedsConsecutiveRegisters(Type *Ty, CallingConv::ID CallConv, bool isVarArg, const DataLayout &DL) const override
Returns true if an argument of type Ty needs to be passed in a contiguous block of registers in calli...
bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override
Return true if folding a constant offset with the given GlobalAddress is legal.
const ARMBaseTargetMachine & getTM() const
bool isMaskAndCmp0FoldingBeneficial(const Instruction &AndI) const override
Return if the target supports combining a chain like:
ShiftLegalizationStrategy preferredShiftLegalizationStrategy(SelectionDAG &DAG, SDNode *N, unsigned ExpansionFactor) const override
bool getPostIndexedAddressParts(SDNode *N, SDNode *Op, SDValue &Base, SDValue &Offset, ISD::MemIndexedMode &AM, SelectionDAG &DAG) const override
getPostIndexedAddressParts - returns true by value, base pointer and offset pointer and addressing mo...
Instruction * emitLeadingFence(IRBuilderBase &Builder, Instruction *Inst, AtomicOrdering Ord) const override
Inserts in the IR a target-specific intrinsic specifying a fence.
bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const override
Return true if Op can create undef or poison from non-undef & non-poison operands.
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
An instruction that atomically checks whether a specified value is in a memory location,...
an instruction that atomically reads a memory location, combines it with another value,...
bool isFloatingPointOperation() const
static LLVM_ABI Attribute get(LLVMContext &Context, AttrKind Kind, uint64_t Val=0)
Return a uniquified Attribute object.
static LLVM_ABI BaseIndexOffset match(const SDNode *N, const SelectionDAG &DAG)
Parses tree in N for base, index, offset addresses.
LLVM Basic Block Representation.
Definition BasicBlock.h:62
The address of a basic block.
Definition Constants.h:1088
static constexpr BranchProbability getZero()
A "pseudo-class" with methods for operating on BUILD_VECTORs.
LLVM_ABI bool isConstantSplat(APInt &SplatValue, APInt &SplatUndef, unsigned &SplatBitSize, bool &HasAnyUndefs, unsigned MinSplatBits=0, bool isBigEndian=false) const
Check if this is a constant splat, and if so, find the smallest element size that splats the vector.
LLVM_ABI int32_t getConstantFPSplatPow2ToLog2Int(BitVector *UndefElements, uint32_t BitWidth) const
If this is a constant FP splat and the splatted constant FP is an exact power or 2,...
CCState - This class holds information needed while lowering arguments and return values.
void getInRegsParamInfo(unsigned InRegsParamRecordIndex, unsigned &BeginReg, unsigned &EndReg) const
unsigned getFirstUnallocated(ArrayRef< MCPhysReg > Regs) const
getFirstUnallocated - Return the index of the first unallocated register in the set,...
static LLVM_ABI bool resultsCompatible(CallingConv::ID CalleeCC, CallingConv::ID CallerCC, MachineFunction &MF, LLVMContext &C, const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn CalleeFn, CCAssignFn CallerFn)
Returns true if the results of the two calling conventions are compatible.
MCRegister AllocateReg(MCPhysReg Reg)
AllocateReg - Attempt to allocate one register.
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
unsigned getInRegsParamsProcessed() const
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
void addInRegsParamInfo(unsigned RegBegin, unsigned RegEnd)
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
unsigned getInRegsParamsCount() const
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
bool needsCustom() const
int64_t getLocMemOffset() const
unsigned getValNo() const
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
LLVM_ABI bool isMustTailCall() const
Tests if this call site must be tail call optimized.
AttributeList getAttributes() const
Return the attributes for this call.
void addParamAttr(unsigned ArgNo, Attribute::AttrKind Kind)
Adds the attribute to the indicated argument.
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
static Constant * get(LLVMContext &Context, ArrayRef< ElementTy > Elts)
get() constructor - Return a constant with array type with an element count and element type matching...
Definition Constants.h:878
const APFloat & getValueAPF() const
ConstantFP - Floating Point Values [float, double].
Definition Constants.h:420
This is the shared class of boolean and integer constants.
Definition Constants.h:87
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
MachineConstantPoolValue * getMachineCPVal() const
const Constant * getConstVal() const
LLVM_ABI Type * getType() const
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
int64_t getSExtValue() const
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
bool isLittleEndian() const
Layout endianness...
Definition DataLayout.h:217
bool isBigEndian() const
Definition DataLayout.h:218
MaybeAlign getStackAlignment() const
Returns the natural stack alignment, or MaybeAlign() if one wasn't specified.
Definition DataLayout.h:250
LLVM_ABI TypeSize getTypeAllocSize(Type *Ty) const
Returns the offset in bytes between successive objects of the specified type, including alignment pad...
StringRef getInternalSymbolPrefix() const
Definition DataLayout.h:308
LLVM_ABI Align getPreferredAlign(const GlobalVariable *GV) const
Returns the preferred alignment of the specified global.
LLVM_ABI Align getPrefTypeAlign(Type *Ty) const
Returns the preferred stack/global alignment for the specified type.
A debug info location.
Definition DebugLoc.h:126
bool empty() const
Definition DenseMap.h:717
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:767
iterator end()
Definition DenseMap.h:687
unsigned size() const
Definition DenseMap.h:718
iterator begin()
Definition DenseMap.h:683
This is a fast-path instruction selection class that generates poor code and doesn't support illegal ...
Definition FastISel.h:67
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
static LLVM_ABI FixedVectorType * get(Type *ElementType, unsigned NumElts)
Definition Type.cpp:843
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
Type * getParamType(unsigned i) const
Parameter type accessors.
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
bool hasMinSize() const
Optimize this function for minimum size (-Oz).
Definition Function.h:696
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
arg_iterator arg_begin()
Definition Function.h:853
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
bool hasStructRetAttr() const
Determine if the function returns a structure through first or second pointer argument.
Definition Function.h:673
const Argument * const_arg_iterator
Definition Function.h:74
bool isVarArg() const
isVarArg - Return true if this function takes a variable number of arguments.
Definition Function.h:230
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
const GlobalValue * getGlobal() const
bool isDSOLocal() const
bool hasExternalWeakLinkage() const
bool hasDLLImportStorageClass() const
Module * getParent()
Get the module that this global value is contained inside of...
bool isStrongDefinitionForLinker() const
Returns true if this global's definition will be the one chosen by the linker.
@ InternalLinkage
Rename collisions when linking (static functions).
Definition GlobalValue.h:60
Common base class shared among various IRBuilders.
Definition IRBuilder.h:111
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2918
LLVM_ABI bool hasAtomicStore() const LLVM_READONLY
Return true if this atomic instruction stores to memory.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
bool isIndexed() const
Return true if this is a pre/post inc/dec load/store.
Tracks which library functions to use for a particular subtarget or function.
const RTLIB::RuntimeLibcallsInfo & getRuntimeLibcallsInfo() const
CallingConv::ID getLibcallImplCallingConv(RTLIB::LibcallImpl Call) const
Get the CallingConv that should be used for the specified libcall.
RTLIB::LibcallImpl getLibcallImpl(RTLIB::Libcall Call) const
Return the lowering's selection of implementation call for Call.
An instruction for reading from memory.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
Describe properties that are true of each instruction in the target description file.
Machine Value Type.
static MVT getFloatingPointVT(unsigned BitWidth)
static auto integer_fixedlen_vector_valuetypes()
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isInteger() const
Return true if this is an integer or a vector integer type.
static LLVM_ABI MVT getVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
static auto integer_valuetypes()
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
bool isScalarInteger() const
Return true if this is an integer, not including vectors.
static MVT getVectorVT(MVT VT, unsigned NumElements)
MVT getVectorElementType() const
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
static MVT getIntegerVT(unsigned BitWidth)
static auto fp_valuetypes()
bool is64BitVector() const
Return true if this is a 64-bit vector type.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
bool isEHPad() const
Returns true if the block is a landing pad.
LLVM_ABI MachineBasicBlock * getFallThrough(bool JumpToFallThrough=true)
Return the fallthrough block if the block can implicitly transfer control to the block after it by fa...
void setCallFrameSize(unsigned N)
Set the call frame size on entry to this basic block.
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI bool canFallThrough()
Return true if the block can implicitly transfer control to the block after it by falling off the end...
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
Instructions::iterator instr_iterator
MachineInstrBundleIterator< MachineInstr, true > reverse_iterator
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
LLVM_ABI instr_iterator erase(instr_iterator I)
Remove an instruction from the instruction list and delete it.
iterator_range< succ_iterator > successors()
iterator_range< pred_iterator > predecessors()
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
LLVM_ABI void moveAfter(MachineBasicBlock *NewBefore)
LLVM_ABI bool isLiveIn(MCRegister Reg, LaneBitmask LaneMask=LaneBitmask::getAll()) const
Return true if the specified register is in the live in set.
void setIsEHPad(bool V=true)
Indicates the block is a landing pad.
The MachineConstantPool class keeps track of constants referenced by a function which must be spilled...
LLVM_ABI unsigned getConstantPoolIndex(const Constant *C, Align Alignment)
getConstantPoolIndex - Create a new entry in the constant pool or return an existing one.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
LLVM_ABI void computeMaxCallFrameSize(MachineFunction &MF, std::vector< MachineBasicBlock::iterator > *FrameSDOps=nullptr)
Computes the maximum size of a callframe.
void setFrameAddressIsTaken(bool T)
void setHasTailCall(bool V=true)
void setReturnAddressIsTaken(bool s)
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
bool hasVAStart() const
Returns true if the function calls the llvm.va_start intrinsic.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
int getFunctionContextIndex() const
Return the index for the function context object.
Properties which a MachineFunction may have at a given point in time.
unsigned getFunctionNumber() const
getFunctionNumber - Return a unique ID for the current function.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineConstantPool * getConstantPool()
getConstantPool - Return the constant pool object for the current function.
const MachineFunctionProperties & getProperties() const
Get the function properties.
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & addExternalSymbol(const char *FnName, unsigned TargetFlags=0) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addConstantPoolIndex(unsigned Idx, int Offset=0, unsigned TargetFlags=0) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addJumpTableIndex(unsigned Idx, unsigned TargetFlags=0) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
MachineInstr * getInstr() const
If conversion operators fail, use this method to get the MachineInstr explicitly.
Representation of each machine instruction.
bool readsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr reads the specified register.
bool definesRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr fully defines the specified register.
MachineOperand * mop_iterator
iterator/begin/end - Iterate over all operands of a machine instruction.
const MachineOperand & getOperand(unsigned i) const
LLVM_ABI unsigned createJumpTableIndex(const std::vector< MachineBasicBlock * > &DestBBs)
createJumpTableIndex - Create a new jump table.
@ EK_Inline
EK_Inline - Jump table entries are emitted inline at their point of use.
@ EK_BlockAddress
EK_BlockAddress - Each entry is a plain address of block, e.g.: .word LBB123.
A description of a memory reference used in the backend.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MONonTemporal
The memory access is non-temporal.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
LLVM_ABI void setIsRenamable(bool Val=true)
bool isReg() const
isReg - Tests if this is a MO_Register operand.
void setIsDead(bool Val=true)
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
static MachineOperand CreateImm(int64_t Val)
Register getReg() const
getReg - Returns the register number.
LLVM_ABI void setIsDef(bool Val=true)
Change a def to a use, or a use to a def.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
This class is used to represent an MLOAD node.
This class is used to represent an MSTORE node.
This SDNode is used for target intrinsics that touch memory and need an associated MachineMemOperand.
This is an abstract virtual class for memory operations.
Align getAlign() const
bool isSimple() const
Returns true if the memory operation is neither atomic or volatile.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
EVT getMemoryVT() const
Return the type of the in-memory value.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
const Triple & getTargetTriple() const
Get the target triple which is a string describing the target host.
Definition Module.h:328
static PointerType * getUnqual(LLVMContext &C)
This constructs an opaque pointer to an object in the default address space (address space zero).
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
const DebugLoc & getDebugLoc() const
Represents one node in the SelectionDAG.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
LLVM_ABI bool isOnlyUserOf(const SDNode *N) const
Return true if this node is the only use of N.
iterator_range< use_iterator > uses()
SDNodeFlags getFlags() const
static bool hasPredecessorHelper(const SDNode *N, SmallPtrSetImpl< const SDNode * > &Visited, SmallVectorImpl< const SDNode * > &Worklist, unsigned int MaxSteps=0, bool TopologicalPrune=false)
Returns true if N is a predecessor of any node in Worklist.
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
bool use_empty() const
Return true if there are no uses of this node.
unsigned getNumValues() const
Return the number of values defined/returned by this operator.
unsigned getNumOperands() const
Return the number of values used by this operation.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
const APInt & getConstantOperandAPInt(unsigned Num) const
Helper method returns the APInt of a ConstantSDNode operand.
bool isPredecessorOf(const SDNode *N) const
Return true if this node is a predecessor of N.
LLVM_ABI bool hasAnyUseOfValue(unsigned Value) const
Return true if there are any use of the indicated value.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
void setCFIType(uint32_t Type)
bool isUndef() const
Returns true if the node type is UNDEF or POISON.
iterator_range< user_iterator > users()
void setFlags(SDNodeFlags NewFlags)
user_iterator user_begin() const
Provide iteration support to walk over all users of an SDNode.
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
const APInt & getConstantOperandAPInt(unsigned i) const
uint64_t getScalarValueSizeInBits() const
unsigned getResNo() const
get the index which selects a specific result in the SDNode
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
unsigned getNumOperands() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getStackArgumentTokenFactor(SDValue Chain)
Compute a TokenFactor to force all the incoming stack arguments to be loaded from the stack.
const TargetSubtargetInfo & getSubtarget() const
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getSplatValue(SDValue V, bool LegalTypes=false)
If V is a splat vector, return its scalar source operand by extracting that element from the source v...
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDNode * getNodeIfExists(unsigned Opcode, SDVTList VTList, ArrayRef< SDValue > Ops, const SDNodeFlags Flags, bool AllowCommute=false)
Get the specified node if it's already available, or else return NULL.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI bool haveNoCommonBitsSet(SDValue A, SDValue B) const
Return true if A and B have no common bits set.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
std::pair< SDValue, SDValue > SplitVectorOperand(const SDNode *N, unsigned OpNo)
Split the node's operand with EXTRACT_SUBVECTOR and return the low/high part.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
const TargetLowering & getTargetLoweringInfo() const
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
LLVM_ABI SDValue getNegative(SDValue Val, const SDLoc &DL, EVT VT)
Create negative operation as (SUB 0, Val).
LLVM_ABI void setNodeMemRefs(MachineSDNode *N, ArrayRef< MachineMemOperand * > NewMemRefs)
Mutate the specified machine node's memory references to the provided list.
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI std::pair< SDValue, SDValue > SplitVector(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the vector with EXTRACT_SUBVECTOR using the provided VTs and return the low/high part.
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
LLVM_ABI MaybeAlign InferPtrAlign(SDValue Ptr) const
Infer alignment of a load / store address.
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI bool isKnownNeverZero(SDValue Op, unsigned Depth=0) const
Test whether the given SDValue is known to contain non-zero value(s).
LLVM_ABI SDValue getExternalSymbol(const char *Sym, EVT VT)
const TargetMachine & getTarget() const
const LibcallLoweringInfo & getLibcalls() const
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI OverflowKind computeOverflowForSignedAdd(SDValue N0, SDValue N1) const
Determine if the result of the signed addition of 2 nodes can overflow.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI SDValue getCondCode(ISD::CondCode Cond)
void addCallSiteInfo(const SDNode *Node, CallSiteInfo &&CallInfo)
Set CallSiteInfo to be associated with Node.
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue CreateStackTemporary(TypeSize Bytes, Align Alignment)
Create a stack temporary based on the size in bytes and the alignment.
SDValue getTargetConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offset=0, unsigned TargetFlags=0)
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI SDValue getMaskedLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Base, SDValue Offset, SDValue Mask, SDValue Src0, EVT MemVT, MachineMemOperand *MMO, ISD::MemIndexedMode AM, ISD::LoadExtType, bool IsExpanding=false)
DenormalMode getDenormalMode(EVT VT) const
Return the current function's default denormal handling kind for the given floating point type.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
LLVM_ABI SDValue getLogicalNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a logical NOT operation as (XOR Val, BooleanOne).
This instruction constructs a fixed permutation of two input vectors.
VectorType * getType() const
Overload to return most specific vector type.
static LLVM_ABI void getShuffleMask(const Constant *Mask, SmallVectorImpl< int > &Result)
Convert the input shuffle mask operand to a vector of integers.
static LLVM_ABI bool isIdentityMask(ArrayRef< int > Mask, int NumSrcElts)
Return true if this shuffle mask chooses elements from exactly one source vector without lane crossin...
This SDNode is used to implement the code generator support for the llvm IR shufflevector instruction...
int getMaskElt(unsigned Idx) const
ArrayRef< int > getMask() const
static LLVM_ABI bool isSplatMask(ArrayRef< int > Mask)
void insert_range(Range &&R)
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
bool empty() const
Definition SmallSet.h:169
bool erase(const T &V)
Definition SmallSet.h:200
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
iterator insert(iterator I, T &&Elt)
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
This class is used to represent ISD::STORE nodes.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
const unsigned char * bytes_end() const
Definition StringRef.h:125
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
constexpr const char * data() const
Get a pointer to the start of the string (which may not be null terminated).
Definition StringRef.h:138
const unsigned char * bytes_begin() const
Definition StringRef.h:122
TargetInstrInfo - Interface to description of machine instruction set.
Provides information about what library functions are available for the current target.
bool isOperationExpand(unsigned Op, EVT VT) const
Return true if the specified operation is illegal on this target or unlikely to be made legal with cu...
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
virtual void finalizeLowering(MachineFunction &MF) const
Execute target specific actions to finalize target lowering.
void setMaxDivRemBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum div/rem the backend supports.
bool PredictableSelectIsExpensive
Tells the code generator that select is more expensive than a branch if the branch is usually predict...
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
ShiftLegalizationStrategy
Return the preferred strategy to legalize tihs SHIFT instruction, with ExpansionFactor being the recu...
void setMinStackArgumentAlignment(Align Alignment)
Set the minimum stack alignment of an argument.
const TargetMachine & getTargetMachine() const
virtual void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const
Inserts necessary declarations for SSP (stack protection) purpose.
void setIndexedMaskedLoadAction(unsigned IdxMode, MVT VT, LegalizeAction Action)
Indicate that the specified indexed masked load does or does not work with the specified type and ind...
void setIndexedLoadAction(ArrayRef< unsigned > IdxModes, MVT VT, LegalizeAction Action)
Indicate that the specified indexed load does or does not work with the specified type and indicate w...
void setPrefLoopAlignment(Align Alignment)
Set the target's preferred loop alignment.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
Sched::Preference getSchedulingPreference() const
Return target scheduling preference.
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
void setIndexedStoreAction(ArrayRef< unsigned > IdxModes, MVT VT, LegalizeAction Action)
Indicate that the specified indexed store does or does not work with the specified type and indicate ...
ExtractSubvectorCost
Enum that specifies how expensive lowering an EXTRACT_SUBVECTOR is.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
virtual unsigned getMaxSupportedInterleaveFactor() const
Get the maximum supported factor for interleaved memory accesses.
void setIndexedMaskedStoreAction(unsigned IdxMode, MVT VT, LegalizeAction Action)
Indicate that the specified indexed masked store does or does not work with the specified type and in...
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
virtual ShiftLegalizationStrategy preferredShiftLegalizationStrategy(SelectionDAG &DAG, SDNode *N, unsigned ExpansionFactor) const
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
virtual bool allowsMemoryAccess(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
Return true if the target supports a memory access of this type for the given address space and align...
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
virtual std::pair< const TargetRegisterClass *, uint8_t > findRepresentativeClass(const TargetRegisterInfo *TRI, MVT VT) const
Return the largest legal super-reg register class of the register class for the specified type and it...
RTLIB::LibcallImpl getLibcallImpl(RTLIB::Libcall Call) const
Get the libcall impl routine name for the specified libcall.
static StringRef getLibcallImplName(RTLIB::LibcallImpl Call)
Get the libcall routine name for the specified libcall implementation.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
LegalizeTypeAction getTypeAction(LLVMContext &Context, EVT VT) const
Return how we should legalize values of this type, either it is already legal (return 'Legal') or we ...
std::vector< ArgListEntry > ArgListTy
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
bool SimplifyDemandedVectorElts(SDValue Op, const APInt &DemandedEltMask, APInt &KnownUndef, APInt &KnownZero, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Vector Op.
void softenSetCCOperands(SelectionDAG &DAG, EVT VT, SDValue &NewLHS, SDValue &NewRHS, ISD::CondCode &CCCode, const SDLoc &DL, const SDValue OldLHS, const SDValue OldRHS) const
Soften the operands of a comparison.
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
bool parametersInCSRMatch(const MachineRegisterInfo &MRI, const uint32_t *CallerPreservedMask, const SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< SDValue > &OutVals) const
Check whether parameters to a call that are passed in callee saved registers are the same as from the...
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
virtual SDValue LowerToTLSEmulatedModel(const GlobalAddressSDNode *GA, SelectionDAG &DAG) const
Lower TLS global address SDNode for target independent emulated TLS model.
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
bool expandDIVREMByConstant(SDNode *N, SmallVectorImpl< SDValue > &Result, EVT HiLoVT, SelectionDAG &DAG, SDValue LL=SDValue(), SDValue LH=SDValue()) const
Attempt to expand an n-bit div/rem/divrem by constant using an n/2-bit algorithm.
bool isPositionIndependent() const
virtual ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const
Examine constraint string and operand type and determine a weight value.
SDValue buildLegalVectorShuffle(EVT VT, const SDLoc &DL, SDValue N0, SDValue N1, MutableArrayRef< int > Mask, SelectionDAG &DAG) const
Tries to build a legal vector shuffle using the provided parameters or equivalent variations.
static ArgListTy getArgListForFunctionType(FunctionType *FuncTy, const AttributeList &FuncAttrs, ArrayRef< SDValue > Ops)
Build a call argument list for FuncTy, taking the argument node values from Ops and the parameter typ...
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
virtual bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0) const
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
TargetLowering(const TargetLowering &)=delete
bool isConstTrueVal(SDValue N) const
Return if the N is a constant or constant vector equal to the true value from getBooleanContents().
virtual ArrayRef< MCPhysReg > getRoundingControlRegisters() const
Returns a 0 terminated array of rounding control registers that can be attached into strict FP call.
virtual bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const
Return true if Op can create undef or poison from non-undef & non-poison operands.
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
void setTypeIdForCallsiteInfo(const CallBase *CB, MachineFunction &MF, MachineFunction::CallSiteInfo &CSInfo) const
Primary interface to the complete machine description for the target machine.
TLSModel::Model getTLSModel(const GlobalValue *GV) const
Returns the TLS model which should be used for the given global variable.
const Triple & getTargetTriple() const
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
virtual const TargetSubtargetInfo * getSubtargetImpl(const Function &) const
Virtual method implemented by subclasses that returns a reference to that target's TargetSubtargetInf...
TargetOptions Options
unsigned EnableFastISel
EnableFastISel - This flag enables fast-path instruction selection which trades away generated code q...
unsigned GuaranteedTailCallOpt
GuaranteedTailCallOpt - This flag is enabled when -tailcallopt is specified on the commandline.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetRegisterInfo * getRegisterInfo() const =0
Return the target's register information.
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
ObjectFormatType getObjectFormat() const
Get the object format for this triple.
Definition Triple.h:539
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isPointerTy() const
True if this is an instance of PointerType.
Definition Type.h:277
bool isFloatTy() const
Return true if this is 'float', a 32-bit IEEE fp type.
Definition Type.h:155
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:272
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
static LLVM_ABI IntegerType * getInt16Ty(LLVMContext &C)
Definition Type.cpp:298
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
LLVMContext & getContext() const
Return the LLVMContext in which this type was uniqued.
Definition Type.h:130
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
bool isFPOrFPVectorTy() const
Return true if this is a FP type or a vector of FP.
Definition Type.h:222
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
User * getUser() const
Returns the User that contains this Use.
Definition Use.h:61
Value * getOperand(unsigned i) const
Definition User.h:207
unsigned getNumOperands() const
Definition User.h:229
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
Base class of all SIMD vector types.
Type * getElementType() const
std::pair< iterator, bool > insert(const ValueT &V)
Definition DenseSet.h:209
bool contains(const_arg_type_t< ValueT > V) const
Check if the set contains the given element.
Definition DenseSet.h:182
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
const ParentTy * getParent() const
Definition ilist_node.h:34
self_iterator getIterator()
Definition ilist_node.h:123
IteratorT end() const
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
static CondCodes getOppositeCondition(CondCodes CC)
Definition ARMBaseInfo.h:49
static ARMCC::CondCodes getSwappedCondition(ARMCC::CondCodes CC)
getSwappedCondition - assume the flags are set by MI(a,b), return the condition code if we modify the...
Definition ARMBaseInfo.h:72
@ SECREL
Thread Pointer Offset.
@ GOT_PREL
Thread Local Storage (General Dynamic Mode)
@ SBREL
Section Relative (Windows TLS)
@ GOTTPOFF
Global Offset Table, PC Relative.
@ TPOFF
Global Offset Table, Thread Pointer Offset.
TOF
Target Operand Flag enum.
@ MO_NONLAZY
MO_NONLAZY - This is an independent flag, on a symbol operand "FOO" it represents a symbol which,...
@ MO_SBREL
MO_SBREL - On a symbol operand, this represents a static base relative relocation.
@ MO_DLLIMPORT
MO_DLLIMPORT - On a symbol operand, this represents that the reference to the symbol is for an import...
@ MO_GOT
MO_GOT - On a symbol operand, this represents a GOT relative relocation.
@ MO_COFFSTUB
MO_COFFSTUB - On a symbol operand "FOO", this indicates that the reference is actually to the "....
static ShiftOpc getShiftOpcForNode(unsigned Opcode)
int getSOImmVal(unsigned Arg)
getSOImmVal - Given a 32-bit immediate, if it is something that can fit into an shifter_operand immed...
int getFP32Imm(const APInt &Imm)
getFP32Imm - Return an 8-bit floating-point version of the 32-bit floating-point value.
uint64_t decodeVMOVModImm(unsigned ModImm, unsigned &EltBits)
decodeVMOVModImm - Decode a NEON/MVE modified immediate value into the element value and the element ...
unsigned getAM2Offset(unsigned AM2Opc)
bool isThumbImmShiftedVal(unsigned V)
isThumbImmShiftedVal - Return true if the specified value can be obtained by left shifting a 8-bit im...
int getT2SOImmVal(unsigned Arg)
getT2SOImmVal - Given a 32-bit immediate, if it is something that can fit into a Thumb-2 shifter_oper...
unsigned createVMOVModImm(unsigned OpCmode, unsigned Val)
int getFP64Imm(const APInt &Imm)
getFP64Imm - Return an 8-bit floating-point version of the 64-bit floating-point value.
int getFP16Imm(const APInt &Imm)
getFP16Imm - Return an 8-bit floating-point version of the 16-bit floating-point value.
unsigned getSORegOpc(ShiftOpc ShOp, unsigned Imm)
int getFP32FP16Imm(const APInt &Imm)
If this is a FP16Imm encoded as a fp32 value, return the 8-bit encoding for it.
AddrOpc getAM2Op(unsigned AM2Opc)
bool isBitFieldInvertedMask(unsigned v)
const unsigned FPStatusBits
FastISel * createFastISel(FunctionLoweringInfo &funcInfo, const TargetLibraryInfo *libInfo, const LibcallLoweringInfo *libcallLowering)
const unsigned FPReservedBits
const unsigned RoundingBitsPos
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ Entry
Definition COFF.h:862
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ Swift
Calling convention for Swift.
Definition CallingConv.h:69
@ ARM_APCS
ARM Procedure Calling Standard (obsolete, but still used on some targets).
@ CFGuard_Check
Special calling convention on Windows for calling the Control Guard Check ICall funtion.
Definition CallingConv.h:82
@ PreserveMost
Used for runtime calls that preserves most registers.
Definition CallingConv.h:63
@ ARM_AAPCS
ARM Architecture Procedure Calling Standard calling convention (aka EABI).
@ CXX_FAST_TLS
Used for access functions.
Definition CallingConv.h:72
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ PreserveAll
Used for runtime calls that preserves (almost) all registers.
Definition CallingConv.h:66
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ Tail
Attemps to make calls as fast as possible while guaranteeing that tail call optimization can always b...
Definition CallingConv.h:76
@ SwiftTail
This follows the Swift calling convention in how arguments are passed but guarantees tail calls will ...
Definition CallingConv.h:87
@ ARM_AAPCS_VFP
Same as ARM_AAPCS, but uses hard floating point ABI.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:43
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
Definition ISDOpcodes.h:263
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ STRICT_FSETCC
STRICT_FSETCC/STRICT_FSETCCS - Constrained versions of SETCC, used for floating-point operands only.
Definition ISDOpcodes.h:516
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ SET_FPENV
Sets the current floating-point environment.
@ MLOAD
Masked load and store - consecutive vector load and store operations with additional mask operand tha...
@ EH_SJLJ_LONGJMP
OUTCHAIN = EH_SJLJ_LONGJMP(INCHAIN, buffer) This corresponds to the eh.sjlj.longjmp intrinsic.
Definition ISDOpcodes.h:170
@ FGETSIGN
INT = FGETSIGN(FP) - Return the sign bit of the specified floating point value as an integer 0/1 valu...
Definition ISDOpcodes.h:543
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ RESET_FPENV
Set floating-point environment to default state.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ SET_FPMODE
Sets the current dynamic floating-point control modes.
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ FMODF
FMODF - Decomposes the operand into integral and fractional parts, each having the same type and sign...
@ FATAN2
FATAN2 - atan2, inspired by libm.
@ FSINCOSPI
FSINCOSPI - Compute both the sine and cosine times pi more accurately than FSINCOS(pi*x),...
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ EH_SJLJ_SETUP_DISPATCH
OUTCHAIN = EH_SJLJ_SETUP_DISPATCH(INCHAIN) The target initializes the dispatch table here.
Definition ISDOpcodes.h:174
@ GlobalAddress
Definition ISDOpcodes.h:90
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:589
@ VECREDUCE_FMAX
FMIN/FMAX nodes can have flags, for NaN/NoNaN variants.
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ ATOMIC_FENCE
OUTCHAIN = ATOMIC_FENCE(INCHAIN, ordering, scope) This corresponds to the fence instruction.
@ RESET_FPMODE
Sets default dynamic floating-point control modes.
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ STRICT_FSQRT
Constrained versions of libm-equivalent floating point intrinsics.
Definition ISDOpcodes.h:441
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:91
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ SET_ROUNDING
Set rounding mode.
Definition ISDOpcodes.h:993
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ AVGCEILS
AVGCEILS/AVGCEILU - Rounding averaging add - Add two integers using an integer of type i[N+2],...
Definition ISDOpcodes.h:725
@ STRICT_UINT_TO_FP
Definition ISDOpcodes.h:490
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ BR
Control flow instructions. These all have token chains.
@ VECREDUCE_FADD
These reductions have relaxed evaluation order semantics, and have a single vector operand.
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ SETCCCARRY
Like SetCC, ops #0 and #1 are the LHS and RHS operands to compare, but op #2 is a boolean indicating ...
Definition ISDOpcodes.h:845
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:355
@ BR_JT
BR_JT - Jumptable branch.
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:249
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:351
@ CTLS
Count leading redundant sign bits.
Definition ISDOpcodes.h:810
@ VECREDUCE_ADD
Integer reductions may have a result type larger than the vector element type.
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
Definition ISDOpcodes.h:988
@ STRICT_FP_TO_FP16
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ GET_FPMODE
Reads the current dynamic floating-point control modes.
@ STRICT_FP16_TO_FP
@ GET_FPENV
Gets the current floating-point environment.
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ READ_REGISTER
READ_REGISTER, WRITE_REGISTER - This node represents llvm.register on the DAG, which implements the n...
Definition ISDOpcodes.h:141
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
Definition ISDOpcodes.h:359
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ STRICT_SINT_TO_FP
STRICT_[US]INT_TO_FP - Convert a signed or unsigned integer to a floating point value.
Definition ISDOpcodes.h:489
@ STRICT_FROUNDEVEN
Definition ISDOpcodes.h:469
@ BF16_TO_FP
BF16_TO_FP, FP_TO_BF16 - These operators are used to perform promotions and truncation for bfloat16.
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:112
@ STRICT_FP_TO_UINT
Definition ISDOpcodes.h:483
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
Definition ISDOpcodes.h:505
@ STRICT_FP_TO_SINT
STRICT_FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:482
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ READCYCLECOUNTER
READCYCLECOUNTER - This corresponds to the readcyclecounter intrinsic.
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:510
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ SCMP
[US]CMP - 3-way comparison of signed or unsigned integers.
Definition ISDOpcodes.h:745
@ AVGFLOORS
AVGFLOORS/AVGFLOORU - Averaging add - Add two integers using an integer of type i[N+1],...
Definition ISDOpcodes.h:720
@ STRICT_FADD
Constrained versions of the binary floating point operators.
Definition ISDOpcodes.h:430
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:805
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ SPONENTRY
SPONENTRY - Represents the llvm.sponentry intrinsic.
Definition ISDOpcodes.h:124
@ STRICT_FNEARBYINT
Definition ISDOpcodes.h:461
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:963
@ EH_SJLJ_SETJMP
RESULT, OUTCHAIN = EH_SJLJ_SETJMP(INCHAIN, buffer) This corresponds to the eh.sjlj....
Definition ISDOpcodes.h:164
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ VAARG
VAARG - VAARG has four operands: an input chain, a pointer, a SRCVALUE, and the alignment.
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
@ ABDS
ABDS/ABDU - Absolute difference - Return the absolute difference between two numbers interpreted as s...
Definition ISDOpcodes.h:732
@ SADDO_CARRY
Carry-using overflow-aware nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:341
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
bool isZEXTLoad(const SDNode *N)
Returns true if the specified node is a ZEXTLOAD.
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
bool isEXTLoad(const SDNode *N)
Returns true if the specified node is a EXTLOAD.
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
bool isSignedIntSetCC(CondCode Code)
Return true if this is a setcc instruction that performs a signed comparison when used with integer o...
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
MemIndexedMode
MemIndexedMode enum - This enum defines the load / store indexed addressing modes.
bool isSEXTLoad(const SDNode *N)
Returns true if the specified node is a SEXTLOAD.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
static const int LAST_INDEXED_MODE
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getOrInsertDeclaration(Module *M, ID id, ArrayRef< Type * > OverloadTys={})
Look up the Function declaration of the intrinsic id in the Module M.
LLVM_ABI Libcall getSINTTOFP(EVT OpVT, EVT RetVT)
getSINTTOFP - Return the SINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getUINTTOFP(EVT OpVT, EVT RetVT)
getUINTTOFP - Return the UINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPTOUINT(EVT OpVT, EVT RetVT)
getFPTOUINT - Return the FPTOUINT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPTOSINT(EVT OpVT, EVT RetVT)
getFPTOSINT - Return the FPTOSINT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPEXT(EVT OpVT, EVT RetVT)
getFPEXT - Return the FPEXT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPROUND(EVT OpVT, EVT RetVT)
getFPROUND - Return the FPROUND_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
@ SingleThread
Synchronized with respect to signal handlers executing in the same thread.
Definition LLVMContext.h:55
initializer< Ty > init(const Ty &Val)
BaseReg
Stack frame base register. Bit 0 of FREInfo.Info.
Definition SFrame.h:77
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
void dump(const SparseBitVector< ElementSize > &LHS, raw_ostream &out)
bool RetFastCC_ARM_APCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
@ Length
Definition DWP.cpp:577
void stable_sort(R &&Range)
Definition STLExtras.h:2132
bool isVTRN_v_undef_Mask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
isVTRN_v_undef_Mask - Special case of isVTRNMask for canonical form of "vector_shuffle v,...
auto find(R &&Range, const T &Val)
Provide wrappers to std::find which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1781
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
bool HasLowerConstantMaterializationCost(unsigned Val1, unsigned Val2, const ARMSubtarget *Subtarget, bool ForCodesize=false)
Returns true if Val1 has a lower Constant Materialization Cost than Val2.
bool isVZIPMask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
InstructionCost Cost
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
RelativeUniformCounterPtr Values
Definition InstrProf.h:91
@ Known
Known to have no common set bits.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
@ Define
Register definition.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isStrongerThanMonotonic(AtomicOrdering AO)
int countr_one(T Value)
Count the number of ones from the least significant bit to the first zero bit.
Definition bit.h:315
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
bool CC_ARM_AAPCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
constexpr bool isMask_32(uint32_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:256
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
bool RetCC_ARM_AAPCS_VFP(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
bool RetCC_ARM_APCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
int bit_width(T Value)
Returns the number of bits needed to represent Value if Value is nonzero.
Definition bit.h:325
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
bool RetCC_ARM_AAPCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
LLVM_ABI Value * concatenateVectors(IRBuilderBase &Builder, ArrayRef< Value * > Vecs)
Concatenate a list of vectors.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
void shuffle(Iterator first, Iterator last, RNG &&g)
Definition STLExtras.h:1546
bool CC_ARM_APCS_GHC(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
static std::array< MachineOperand, 2 > predOps(ARMCC::CondCodes Pred, unsigned PredReg=0)
Get the operands corresponding to the given Pred value.
bool operator==(const AddressRangeValuePair &LHS, const AddressRangeValuePair &RHS)
constexpr bool isShiftedMask_32(uint32_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (32 bit ver...
Definition MathExtras.h:268
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
bool isReleaseOrStronger(AtomicOrdering AO)
auto dyn_cast_or_null(const Y &Val)
Definition Casting.h:753
constexpr bool has_single_bit(T Value) noexcept
Definition bit.h:149
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
LLVM_ABI bool isBitwiseNot(SDValue V, bool AllowUndefs=false)
Returns true if V is a bitwise not operation.
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool FastCC_ARM_APCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
bool CC_ARM_Win32_CFGuard_Check(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
SmallVector< ValueTypeFromRangeType< R >, Size > to_vector(R &&Range)
Given a range of type R, iterate the entire range and return a SmallVector with elements of the vecto...
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isVUZPMask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
bool isVTRNMask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
const unsigned PerfectShuffleTable[6561+1]
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Other
Any other memory.
Definition ModRef.h:68
CombineLevel
Definition DAGCombine.h:15
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
unsigned ConstantMaterializationCost(unsigned Val, const ARMSubtarget *Subtarget, bool ForCodesize=false)
Returns the number of instructions required to materialize the given constant in a register,...
@ Mul
Product of integers.
@ And
Bitwise or logical AND of integers.
@ Sub
Subtraction of integers.
@ Add
Sum of integers.
@ FAdd
Sum of floats.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
DWARFExpression::Operation Op
bool isVUZP_v_undef_Mask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
isVUZP_v_undef_Mask - Special case of isVUZPMask for canonical form of "vector_shuffle v,...
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr U AbsoluteValue(T X)
Return the absolute value of a signed integer, converted to the corresponding unsigned integer type.
Definition MathExtras.h:587
bool isAcquireOrStronger(AtomicOrdering AO)
constexpr unsigned BitWidth
ExceptionHandling
Definition CodeGen.h:54
@ SjLj
setjmp/longjmp based exceptions
Definition CodeGen.h:58
static MachineOperand t1CondCodeOp(bool isDead=false)
Get the operand corresponding to the conditional code result for Thumb1.
auto count_if(R &&Range, UnaryPredicate P)
Wrapper function around std::count_if to count the number of times an element satisfying a given pred...
Definition STLExtras.h:2035
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
constexpr auto seq(T Begin, T End)
Iterate over an integral type from Begin up to - but not including - End.
Definition Sequence.h:341
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
constexpr bool isIntN(unsigned N, int64_t x)
Checks if an signed integer fits into the given (dynamic) bit width.
Definition MathExtras.h:249
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
static MachineOperand condCodeOp(unsigned CCReg=0)
Get the operand corresponding to the conditional code result.
bool isVREVMask(ArrayRef< int > M, EVT VT, unsigned BlockSize)
isVREVMask - Check if a vector shuffle corresponds to a VREV instruction with the specified blocksize...
unsigned gettBLXrOpcode(const MachineFunction &MF)
bool CC_ARM_APCS(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
@ Increment
Incrementally increasing token ID.
Definition AllocToken.h:26
bool CC_ARM_AAPCS_VFP(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
LLVM_ABI llvm::SmallVector< int, 16 > createSequentialMask(unsigned Start, unsigned NumInts, unsigned NumUndefs)
Create a sequential shuffle mask.
constexpr bool isShiftedUInt(uint64_t x)
Checks if a unsigned integer is an N bit number shifted left by S.
Definition MathExtras.h:199
unsigned convertAddSubFlagsOpcode(unsigned OldOpc)
Map pseudo instructions that imply an 'S' bit onto real opcodes.
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
bool isVZIP_v_undef_Mask(ArrayRef< int > M, EVT VT, unsigned &WhichResult)
isVZIP_v_undef_Mask - Special case of isVZIPMask for canonical form of "vector_shuffle v,...
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Load/store instruction that can be merged with a base address update.
SDNode * N
Instruction that updates a pointer.
unsigned ConstInc
Pointer increment value if it is a constant, or 0 otherwise.
SDValue Inc
Pointer increment operand.
A collection of metadata nodes that might be associated with a memory access used by the alias-analys...
Definition Metadata.h:774
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static constexpr DenormalMode getIEEE()
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
EVT getDoubleNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:494
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
Definition ValueTypes.h:382
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
Definition ValueTypes.h:501
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isFixedLengthVector() const
Definition ValueTypes.h:199
static EVT getFloatingPointVT(unsigned BitWidth)
Returns the EVT that represents a floating-point type with the given number of bits.
Definition ValueTypes.h:55
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool bitsLE(EVT VT) const
Return true if this has no more bits than VT.
Definition ValueTypes.h:331
EVT getHalfNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:484
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
bool is64BitVector() const
Return true if this is a 64-bit vector type.
Definition ValueTypes.h:225
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
EVT ArgVT
Usually the non-legalized type of the argument, which is the EVT corresponding to the OrigTy IR type.
static KnownBits makeConstant(const APInt &C)
Create known bits from a known constant.
Definition KnownBits.h:315
unsigned getBitWidth() const
Get the bit width of this value.
Definition KnownBits.h:44
KnownBits zext(unsigned BitWidth) const
Return known bits for a zero extension of the value we're tracking.
Definition KnownBits.h:176
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
static LLVM_ABI KnownBits mul(const KnownBits &LHS, const KnownBits &RHS, bool NoUndefSelfMultiply=false)
Compute known bits resulting from multiplying LHS and RHS.
APInt getSignedMinValue() const
Return the minimal signed value possible given these KnownBits.
Definition KnownBits.h:136
Matching combinators.
SmallVector< ArgRegPair, 1 > ArgRegPairs
Vector of call argument and its forwarding register.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getJumpTable(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a jump table entry.
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
LLVM_ABI std::pair< FunctionType *, AttributeList > getFunctionTy(LLVMContext &Ctx, const Triple &TT, const DataLayout &DL, RTLIB::LibcallImpl LibcallImpl) const
These are IR-level optimization flags that may be propagated to SDNodes.
bool hasNoSignedZeros() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
This contains information for each constraint that we are lowering.
This structure contains all information that is necessary for lowering calls.
CallLoweringInfo & setInRegister(bool Value=true)
CallLoweringInfo & setLibCallee(CallingConv::ID CC, Type *ResultType, SDValue Target, ArgListTy &&ArgsList)
SmallVector< ISD::InputArg, 32 > Ins
CallLoweringInfo & setZExtResult(bool Value=true)
CallLoweringInfo & setDebugLoc(const SDLoc &dl)
CallLoweringInfo & setSExtResult(bool Value=true)
SmallVector< ISD::OutputArg, 32 > Outs
CallLoweringInfo & setChain(SDValue InChain)
CallLoweringInfo & setCallee(CallingConv::ID CC, Type *ResultType, SDValue Target, ArgListTy &&ArgsList, AttributeSet ResultAttrs={})
LLVM_ABI void AddToWorklist(SDNode *N)
LLVM_ABI SDValue CombineTo(SDNode *N, ArrayRef< SDValue > To, bool AddTo=true)
This structure is used to pass arguments to makeLibCall function.
A convenience struct that encapsulates a DAG, and two SDValues for returning information from TargetL...