LLVM 24.0.0git
AArch64ISelDAGToDAG.cpp
Go to the documentation of this file.
1//===-- AArch64ISelDAGToDAG.cpp - A dag to dag inst selector for AArch64 --===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file defines an instruction selector for the AArch64 target.
10//
11//===----------------------------------------------------------------------===//
12
13#include "AArch64.h"
14#include "AArch64ExpandImm.h"
18#include "llvm/ADT/APSInt.h"
22#include "llvm/IR/Function.h" // To access function attributes.
23#include "llvm/IR/GlobalValue.h"
24#include "llvm/IR/Intrinsics.h"
25#include "llvm/IR/IntrinsicsAArch64.h"
27#include "llvm/Support/Debug.h"
32
33using namespace llvm;
34using namespace llvm::SDPatternMatch;
35
36#define DEBUG_TYPE "aarch64-isel"
37#define PASS_NAME "AArch64 Instruction Selection"
38
39// https://github.com/llvm/llvm-project/issues/114425
40#if defined(_MSC_VER) && !defined(__clang__) && !defined(NDEBUG)
41#pragma inline_depth(0)
42#endif
43
44//===--------------------------------------------------------------------===//
45/// AArch64DAGToDAGISel - AArch64 specific code to select AArch64 machine
46/// instructions for SelectionDAG operations.
47///
48namespace {
49
50class AArch64DAGToDAGISel : public SelectionDAGISel {
51
52 /// Subtarget - Keep a pointer to the AArch64Subtarget around so that we can
53 /// make the right decision when generating code for different targets.
54 const AArch64Subtarget *Subtarget;
55
56public:
57 AArch64DAGToDAGISel() = delete;
58
59 explicit AArch64DAGToDAGISel(AArch64TargetMachine &tm,
60 CodeGenOptLevel OptLevel)
61 : SelectionDAGISel(tm, OptLevel), Subtarget(nullptr) {}
62
63 bool runOnMachineFunction(MachineFunction &MF) override {
64 Subtarget = &MF.getSubtarget<AArch64Subtarget>();
66 }
67
68 void Select(SDNode *Node) override;
69 void PreprocessISelDAG() override;
70
71 /// SelectInlineAsmMemoryOperand - Implement addressing mode selection for
72 /// inline asm expressions.
73 bool SelectInlineAsmMemoryOperand(const SDValue &Op,
74 InlineAsm::ConstraintCode ConstraintID,
75 std::vector<SDValue> &OutOps) override;
76
77 template <signed Low, signed High, signed Scale>
78 bool SelectRDVLImm(SDValue N, SDValue &Imm);
79
80 template <signed Low, signed High>
81 bool SelectRDSVLShiftImm(SDValue N, SDValue &Imm);
82
83 bool SelectArithExtendedRegister(SDValue N, SDValue &Reg, SDValue &Shift);
84 bool SelectArithUXTXRegister(SDValue N, SDValue &Reg, SDValue &Shift);
85 bool SelectArithImmed(SDValue N, SDValue &Val, SDValue &Shift);
86 bool SelectNegArithImmed(SDValue N, SDValue &Val, SDValue &Shift);
87 bool SelectArithShiftedRegister(SDValue N, SDValue &Reg, SDValue &Shift) {
88 return SelectShiftedRegister(N, false, Reg, Shift);
89 }
90 bool SelectLogicalShiftedRegister(SDValue N, SDValue &Reg, SDValue &Shift) {
91 return SelectShiftedRegister(N, true, Reg, Shift);
92 }
93 template <unsigned ShiftWidth>
94 bool SelectShiftMask(SDValue N, SDValue &ShAmt);
95
96 bool SelectAddrModeIndexed7S8(SDValue N, SDValue &Base, SDValue &OffImm) {
97 return SelectAddrModeIndexed7S(N, 1, Base, OffImm);
98 }
99 bool SelectAddrModeIndexed7S16(SDValue N, SDValue &Base, SDValue &OffImm) {
100 return SelectAddrModeIndexed7S(N, 2, Base, OffImm);
101 }
102 bool SelectAddrModeIndexed7S32(SDValue N, SDValue &Base, SDValue &OffImm) {
103 return SelectAddrModeIndexed7S(N, 4, Base, OffImm);
104 }
105 bool SelectAddrModeIndexed7S64(SDValue N, SDValue &Base, SDValue &OffImm) {
106 return SelectAddrModeIndexed7S(N, 8, Base, OffImm);
107 }
108 bool SelectAddrModeIndexed7S128(SDValue N, SDValue &Base, SDValue &OffImm) {
109 return SelectAddrModeIndexed7S(N, 16, Base, OffImm);
110 }
111 bool SelectAddrModeIndexedS9S128(SDValue N, SDValue &Base, SDValue &OffImm) {
112 return SelectAddrModeIndexedBitWidth(N, true, 9, 16, Base, OffImm);
113 }
114 bool SelectAddrModeIndexedU6S128(SDValue N, SDValue &Base, SDValue &OffImm) {
115 return SelectAddrModeIndexedBitWidth(N, false, 6, 16, Base, OffImm);
116 }
117 bool SelectAddrModeIndexed8(SDValue N, SDValue &Base, SDValue &OffImm) {
118 return SelectAddrModeIndexed(N, 1, Base, OffImm);
119 }
120 bool SelectAddrModeIndexed16(SDValue N, SDValue &Base, SDValue &OffImm) {
121 return SelectAddrModeIndexed(N, 2, Base, OffImm);
122 }
123 bool SelectAddrModeIndexed32(SDValue N, SDValue &Base, SDValue &OffImm) {
124 return SelectAddrModeIndexed(N, 4, Base, OffImm);
125 }
126 bool SelectAddrModeIndexed64(SDValue N, SDValue &Base, SDValue &OffImm) {
127 return SelectAddrModeIndexed(N, 8, Base, OffImm);
128 }
129 bool SelectAddrModeIndexed128(SDValue N, SDValue &Base, SDValue &OffImm) {
130 return SelectAddrModeIndexed(N, 16, Base, OffImm);
131 }
132 bool SelectAddrModeUnscaled8(SDValue N, SDValue &Base, SDValue &OffImm) {
133 return SelectAddrModeUnscaled(N, 1, Base, OffImm);
134 }
135 bool SelectAddrModeUnscaled16(SDValue N, SDValue &Base, SDValue &OffImm) {
136 return SelectAddrModeUnscaled(N, 2, Base, OffImm);
137 }
138 bool SelectAddrModeUnscaled32(SDValue N, SDValue &Base, SDValue &OffImm) {
139 return SelectAddrModeUnscaled(N, 4, Base, OffImm);
140 }
141 bool SelectAddrModeUnscaled64(SDValue N, SDValue &Base, SDValue &OffImm) {
142 return SelectAddrModeUnscaled(N, 8, Base, OffImm);
143 }
144 bool SelectAddrModeUnscaled128(SDValue N, SDValue &Base, SDValue &OffImm) {
145 return SelectAddrModeUnscaled(N, 16, Base, OffImm);
146 }
147 template <unsigned Size, unsigned Max>
148 bool SelectAddrModeIndexedUImm(SDValue N, SDValue &Base, SDValue &OffImm) {
149 // Test if there is an appropriate addressing mode and check if the
150 // immediate fits.
151 bool Found = SelectAddrModeIndexed(N, Size, Base, OffImm);
152 if (Found) {
153 if (auto *CI = dyn_cast<ConstantSDNode>(OffImm)) {
154 int64_t C = CI->getSExtValue();
155 if (C <= Max)
156 return true;
157 }
158 }
159
160 // Otherwise, base only, materialize address in register.
161 Base = N;
162 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
163 return true;
164 }
165
166 template<int Width>
167 bool SelectAddrModeWRO(SDValue N, SDValue &Base, SDValue &Offset,
168 SDValue &SignExtend, SDValue &DoShift) {
169 return SelectAddrModeWRO(N, Width / 8, Base, Offset, SignExtend, DoShift);
170 }
171
172 template<int Width>
173 bool SelectAddrModeXRO(SDValue N, SDValue &Base, SDValue &Offset,
174 SDValue &SignExtend, SDValue &DoShift) {
175 return SelectAddrModeXRO(N, Width / 8, Base, Offset, SignExtend, DoShift);
176 }
177
178 bool SelectExtractHigh(SDValue N, SDValue &Res) {
179 if (Subtarget->isLittleEndian() && N->getOpcode() == ISD::BITCAST)
180 N = N->getOperand(0);
181 if (N->getOpcode() != ISD::EXTRACT_SUBVECTOR ||
182 !isa<ConstantSDNode>(N->getOperand(1)))
183 return false;
184 EVT VT = N->getValueType(0);
185 EVT LVT = N->getOperand(0).getValueType();
186 unsigned Index = N->getConstantOperandVal(1);
187 if (!VT.is64BitVector() || !LVT.is128BitVector() ||
188 Index != VT.getVectorNumElements())
189 return false;
190 Res = N->getOperand(0);
191 return true;
192 }
193
194 bool SelectRoundingVLShr(SDValue N, SDValue &Res1, SDValue &Res2) {
195 if (N.getOpcode() != AArch64ISD::VLSHR)
196 return false;
197 SDValue Op = N->getOperand(0);
198 EVT VT = Op.getValueType();
199 unsigned ShtAmt = N->getConstantOperandVal(1);
200 if (ShtAmt > VT.getScalarSizeInBits() / 2 || Op.getOpcode() != ISD::ADD)
201 return false;
202
203 APInt Imm;
204 if (Op.getOperand(1).getOpcode() == AArch64ISD::MOVIshift)
206 Op.getOperand(1).getConstantOperandVal(0)
207 << Op.getOperand(1).getConstantOperandVal(1));
208 else if (Op.getOperand(1).getOpcode() == AArch64ISD::DUP &&
209 isa<ConstantSDNode>(Op.getOperand(1).getOperand(0)))
211 Op.getOperand(1).getConstantOperandVal(0));
212 else
213 return false;
214
215 if (Imm != 1ULL << (ShtAmt - 1))
216 return false;
217
218 Res1 = Op.getOperand(0);
219 Res2 = CurDAG->getTargetConstant(ShtAmt, SDLoc(N), MVT::i32);
220 return true;
221 }
222
223 bool SelectDupZeroOrUndef(SDValue N) {
224 switch(N->getOpcode()) {
225 case ISD::UNDEF:
226 case ISD::POISON:
227 return true;
228 case AArch64ISD::DUP:
229 case ISD::SPLAT_VECTOR: {
230 auto Opnd0 = N->getOperand(0);
231 if (isNullConstant(Opnd0))
232 return true;
233 if (isNullFPConstant(Opnd0))
234 return true;
235 break;
236 }
237 default:
238 break;
239 }
240
241 return false;
242 }
243
244 bool SelectAny(SDValue) { return true; }
245
246 bool SelectDupZero(SDValue N) {
247 switch(N->getOpcode()) {
248 case AArch64ISD::DUP:
249 case ISD::SPLAT_VECTOR: {
250 auto Opnd0 = N->getOperand(0);
251 if (isNullConstant(Opnd0))
252 return true;
253 if (isNullFPConstant(Opnd0))
254 return true;
255 break;
256 }
257 }
258
259 return false;
260 }
261
262 template <MVT::SimpleValueType VT, bool Negate>
263 bool SelectSVEAddSubImm(SDValue N, SDValue &Imm, SDValue &Shift) {
264 return SelectSVEAddSubImm(N, VT, Imm, Shift, Negate);
265 }
266
267 template <MVT::SimpleValueType VT, bool Negate>
268 bool SelectSVEAddSubSSatImm(SDValue N, SDValue &Imm, SDValue &Shift) {
269 return SelectSVEAddSubSSatImm(N, VT, Imm, Shift, Negate);
270 }
271
272 template <MVT::SimpleValueType VT>
273 bool SelectSVECpyDupImm(SDValue N, SDValue &Imm, SDValue &Shift) {
274 return SelectSVECpyDupImm(N, VT, Imm, Shift);
275 }
276
277 template <MVT::SimpleValueType VT, bool Invert = false>
278 bool SelectSVELogicalImm(SDValue N, SDValue &Imm) {
279 return SelectSVELogicalImm(N, VT, Imm, Invert);
280 }
281
282 template <MVT::SimpleValueType VT>
283 bool SelectSVEArithImm(SDValue N, SDValue &Imm) {
284 return SelectSVEArithImm(N, VT, Imm);
285 }
286
287 template <unsigned Low, unsigned High, bool AllowSaturation = false>
288 bool SelectSVEShiftImm(SDValue N, SDValue &Imm) {
289 return SelectSVEShiftImm(N, Low, High, AllowSaturation, Imm);
290 }
291
292 bool SelectSVEShiftSplatImmR(SDValue N, SDValue &Imm) {
293 if (N->getOpcode() != ISD::SPLAT_VECTOR)
294 return false;
295
296 EVT EltVT = N->getValueType(0).getVectorElementType();
297 return SelectSVEShiftImm(N->getOperand(0), /* Low */ 1,
298 /* High */ EltVT.getFixedSizeInBits(),
299 /* AllowSaturation */ true, Imm);
300 }
301
302 // Returns a suitable CNT/INC/DEC/RDVL multiplier to calculate VSCALE*N.
303 template<signed Min, signed Max, signed Scale, bool Shift>
304 bool SelectCntImm(SDValue N, SDValue &Imm) {
306 return false;
307
308 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
309 if (Shift)
310 MulImm = 1LL << MulImm;
311
312 if ((MulImm % std::abs(Scale)) != 0)
313 return false;
314
315 MulImm /= Scale;
316 if ((MulImm >= Min) && (MulImm <= Max)) {
317 Imm = CurDAG->getTargetConstant(MulImm, SDLoc(N), MVT::i32);
318 return true;
319 }
320
321 return false;
322 }
323
324 template <signed Max, signed Scale>
325 bool SelectEXTImm(SDValue N, SDValue &Imm) {
327 return false;
328
329 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
330
331 if (MulImm >= 0 && MulImm <= Max) {
332 MulImm *= Scale;
333 Imm = CurDAG->getTargetConstant(MulImm, SDLoc(N), MVT::i32);
334 return true;
335 }
336
337 return false;
338 }
339
340 template <unsigned BaseReg, unsigned Max>
341 bool ImmToReg(SDValue N, SDValue &Imm) {
342 if (auto *CI = dyn_cast<ConstantSDNode>(N)) {
343 uint64_t C = CI->getZExtValue();
344
345 if (C > Max)
346 return false;
347
348 Imm = CurDAG->getRegister(BaseReg + C, MVT::Other);
349 return true;
350 }
351 return false;
352 }
353
354 /// Form sequences of consecutive 64/128-bit registers for use in NEON
355 /// instructions making use of a vector-list (e.g. ldN, tbl). Vecs must have
356 /// between 1 and 4 elements. If it contains a single element that is returned
357 /// unchanged; otherwise a REG_SEQUENCE value is returned.
360 // Form a sequence of SVE registers for instructions using list of vectors,
361 // e.g. structured loads and stores (ldN, stN).
362 SDValue createZTuple(ArrayRef<SDValue> Vecs);
363
364 // Similar to above, except the register must start at a multiple of the
365 // tuple, e.g. z2 for a 2-tuple, or z8 for a 4-tuple.
366 SDValue createZMulTuple(ArrayRef<SDValue> Regs);
367
368 /// Generic helper for the createDTuple/createQTuple
369 /// functions. Those should almost always be called instead.
370 SDValue createTuple(ArrayRef<SDValue> Vecs, const unsigned RegClassIDs[],
371 const unsigned SubRegs[]);
372
373 void SelectTable(SDNode *N, unsigned NumVecs, unsigned Opc, bool isExt);
374
375 bool tryIndexedLoad(SDNode *N);
376
377 void SelectPtrauthAuth(SDNode *N);
378 void SelectPtrauthResign(SDNode *N);
379 void SelectPtrauthResignWithPC(SDNode *N);
380
381 bool trySelectStackSlotTagP(SDNode *N);
382 void SelectTagP(SDNode *N);
383
384 void SelectLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
385 unsigned SubRegIdx);
386 void SelectPostLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
387 unsigned SubRegIdx);
388 void SelectLoadLane(SDNode *N, unsigned NumVecs, unsigned Opc);
389 void SelectPostLoadLane(SDNode *N, unsigned NumVecs, unsigned Opc);
390 void SelectPredicatedLoad(SDNode *N, unsigned NumVecs, unsigned Scale,
391 unsigned Opc_rr, unsigned Opc_ri,
392 bool IsIntr = false);
393 void SelectContiguousMultiVectorLoad(SDNode *N, unsigned NumVecs,
394 unsigned Scale, unsigned Opc_ri,
395 unsigned Opc_rr);
396 void SelectDestructiveMultiIntrinsic(SDNode *N, unsigned NumVecs,
397 bool IsZmMulti, unsigned Opcode,
398 bool HasPred = false);
399 void SelectPExtPair(SDNode *N, unsigned Opc);
400 void SelectWhilePair(SDNode *N, unsigned Opc);
401 void SelectCVTIntrinsic(SDNode *N, unsigned NumVecs, unsigned Opcode);
402 void SelectCVTIntrinsicFP8(SDNode *N, unsigned NumVecs, unsigned Opcode);
403 void SelectClamp(SDNode *N, unsigned NumVecs, unsigned Opcode);
404 void SelectUnaryMultiIntrinsic(SDNode *N, unsigned NumOutVecs,
405 bool IsTupleInput, unsigned Opc);
406 void SelectFrintFromVT(SDNode *N, unsigned NumVecs, unsigned Opcode);
407
408 template <unsigned MaxIdx, unsigned Scale>
409 void SelectMultiVectorMove(SDNode *N, unsigned NumVecs, unsigned BaseReg,
410 unsigned Op);
411 void SelectMultiVectorMoveZ(SDNode *N, unsigned NumVecs,
412 unsigned Op, unsigned MaxIdx, unsigned Scale,
413 unsigned BaseReg = 0);
414 /// SVE Reg+Imm addressing mode.
415 template <int64_t Min, int64_t Max>
416 bool SelectAddrModeIndexedSVE(SDNode *Root, SDValue N, SDValue &Base,
417 SDValue &OffImm);
418 /// SVE Reg+Reg address mode.
419 template <unsigned Scale>
420 bool SelectSVERegRegAddrMode(SDValue N, SDValue &Base, SDValue &Offset) {
421 return SelectSVERegRegAddrMode(N, Scale, Base, Offset);
422 }
423
424 void SelectMultiVectorLutiLane(SDNode *Node, unsigned NumOutVecs,
425 unsigned Opc, uint32_t MaxImm);
426 void SelectMultiVectorLuti6LaneX4(SDNode *Node, unsigned NumIndexVecs);
427
428 void SelectMultiVectorLuti(SDNode *Node, unsigned NumOutVecs, unsigned Opc,
429 unsigned NumInVecs);
430
431 template <unsigned MaxIdx, unsigned Scale>
432 bool SelectSMETileSlice(SDValue N, SDValue &Vector, SDValue &Offset) {
433 return SelectSMETileSlice(N, MaxIdx, Vector, Offset, Scale);
434 }
435
436 void SelectStore(SDNode *N, unsigned NumVecs, unsigned Opc);
437 void SelectPostStore(SDNode *N, unsigned NumVecs, unsigned Opc);
438 void SelectStoreLane(SDNode *N, unsigned NumVecs, unsigned Opc);
439 void SelectPostStoreLane(SDNode *N, unsigned NumVecs, unsigned Opc);
440 void SelectPredicatedStore(SDNode *N, unsigned NumVecs, unsigned Scale,
441 unsigned Opc_rr, unsigned Opc_ri);
442 std::tuple<unsigned, SDValue, SDValue>
443 findAddrModeSVELoadStore(SDNode *N, unsigned Opc_rr, unsigned Opc_ri,
444 const SDValue &OldBase, const SDValue &OldOffset,
445 unsigned Scale);
446
447 bool tryBitfieldExtractOp(SDNode *N);
448 bool tryBitfieldInsertOp(SDNode *N);
449 bool tryBitfieldInsertInZeroOp(SDNode *N);
450 bool tryShiftAmountMod(SDNode *N);
451
452 bool tryReadRegister(SDNode *N);
453 bool tryWriteRegister(SDNode *N);
454
455 bool trySelectCastFixedLengthToScalableVector(SDNode *N);
456 bool trySelectCastScalableToFixedLengthVector(SDNode *N);
457
458 bool trySelectXAR(SDNode *N);
459
460 bool tryFoldCselToFMaxMin(SDNode *N);
461
462// Include the pieces autogenerated from the target description.
463#include "AArch64GenDAGISel.inc"
464
465private:
466 bool SelectShiftedRegister(SDValue N, bool AllowROR, SDValue &Reg,
467 SDValue &Shift);
468 bool SelectShiftedRegisterFromAnd(SDValue N, SDValue &Reg, SDValue &Shift);
469 bool SelectAddrModeIndexed7S(SDValue N, unsigned Size, SDValue &Base,
470 SDValue &OffImm) {
471 return SelectAddrModeIndexedBitWidth(N, true, 7, Size, Base, OffImm);
472 }
473 bool SelectAddrModeIndexedBitWidth(SDValue N, bool IsSignedImm, unsigned BW,
474 unsigned Size, SDValue &Base,
475 SDValue &OffImm);
476 bool SelectAddrModeIndexed(SDValue N, unsigned Size, SDValue &Base,
477 SDValue &OffImm);
478 bool SelectAddrModeUnscaled(SDValue N, unsigned Size, SDValue &Base,
479 SDValue &OffImm);
480 bool SelectAddrModeWRO(SDValue N, unsigned Size, SDValue &Base,
481 SDValue &Offset, SDValue &SignExtend,
482 SDValue &DoShift);
483 bool SelectAddrModeXRO(SDValue N, unsigned Size, SDValue &Base,
484 SDValue &Offset, SDValue &SignExtend,
485 SDValue &DoShift);
486 bool isWorthNegatingImm(SDValue V) const;
487 bool isWorthFoldingALU(SDValue V, bool LSL = false) const;
488 bool isWorthFoldingAddr(SDValue V, unsigned Size) const;
489 bool SelectExtendedSHL(SDValue N, unsigned Size, bool WantExtend,
490 SDValue &Offset, SDValue &SignExtend);
491
492 template<unsigned RegWidth>
493 bool SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos) {
494 return SelectCVTFixedPosOperand(N, FixedPos, RegWidth);
495 }
496 bool SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos, unsigned Width);
497
498 template <unsigned RegWidth>
499 bool SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos) {
500 return SelectCVTFixedPointVec(N, FixedPos, RegWidth);
501 }
502 bool SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos, unsigned Width);
503
504 template<unsigned RegWidth>
505 bool SelectCVTFixedPosRecipOperand(SDValue N, SDValue &FixedPos) {
506 return SelectCVTFixedPosRecipOperand(N, FixedPos, RegWidth);
507 }
508
509 bool SelectCVTFixedPosRecipOperand(SDValue N, SDValue &FixedPos,
510 unsigned Width);
511
512 template <unsigned FloatWidth>
513 bool SelectCVTFixedPosRecipOperandVec(SDValue N, SDValue &FixedPos) {
514 return SelectCVTFixedPosRecipOperandVec(N, FixedPos, FloatWidth);
515 }
516
517 bool SelectCVTFixedPosRecipOperandVec(SDValue N, SDValue &FixedPos,
518 unsigned Width);
519
520 bool SelectCMP_SWAP(SDNode *N);
521
522 AArch64MemoryHint decodeMemoryHintFlags(MachineMemOperand *MMO) const;
523 bool isAtomicMemoryHint(SDNode *N, AArch64MemoryHint Hint) const;
524
525 bool SelectSVEAddSubImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift,
526 bool Negate);
527 bool SelectSVEAddSubImm(SDLoc DL, APInt Value, MVT VT, SDValue &Imm,
528 SDValue &Shift, bool Negate);
529 bool SelectSVEAddSubSSatImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift,
530 bool Negate);
531 bool SelectSVECpyDupImm(SDValue N, MVT VT, SDValue &Imm, SDValue &Shift);
532 bool SelectSVELogicalImm(SDValue N, MVT VT, SDValue &Imm, bool Invert);
533
534 // Match `<NEON Splat> SVEImm` (where <NEON Splat> could be fmov, movi, etc).
535 bool SelectNEONSplatOfSVELogicalImm(SDValue N, SDValue &Imm);
536 bool SelectNEONSplatOfSVEAddSubImm(SDValue N, SDValue &Imm, SDValue &Shift);
537 bool SelectNEONSplatOfSVEArithSImm(SDValue N, SDValue &Imm);
538 bool SelectNEONSplatOfSImm8(SDValue N, SDValue &Imm);
539 bool SelectNEONSplatOfUImm8(SDValue N, SDValue &Imm);
540
541 bool SelectSVESignedArithImm(SDLoc DL, APInt Value, SDValue &Imm);
542 bool SelectSVESignedArithImm(SDValue N, SDValue &Imm);
543 bool SelectSVEShiftImm(SDValue N, uint64_t Low, uint64_t High,
544 bool AllowSaturation, SDValue &Imm);
545
546 bool SelectSVEArithImm(SDValue N, MVT VT, SDValue &Imm);
547 bool SelectSVERegRegAddrMode(SDValue N, unsigned Scale, SDValue &Base,
548 SDValue &Offset);
549 bool SelectSMETileSlice(SDValue N, unsigned MaxSize, SDValue &Vector,
550 SDValue &Offset, unsigned Scale = 1);
551
552 bool SelectAllActivePredicate(SDValue N);
553 bool SelectAnyPredicate(SDValue N);
554
555 bool SelectCmpBranchUImm6Operand(SDNode *P, SDValue N, SDValue &Imm);
556
557 template <bool MatchCBB>
558 bool SelectCmpBranchExtOperand(SDValue N, SDValue &Reg, SDValue &ExtType);
559};
560
561class AArch64DAGToDAGISelLegacy : public SelectionDAGISelLegacy {
562public:
563 static char ID;
564 explicit AArch64DAGToDAGISelLegacy(AArch64TargetMachine &tm,
565 CodeGenOptLevel OptLevel)
567 ID, std::make_unique<AArch64DAGToDAGISel>(tm, OptLevel)) {}
568};
569} // end anonymous namespace
570
571char AArch64DAGToDAGISelLegacy::ID = 0;
572
573INITIALIZE_PASS(AArch64DAGToDAGISelLegacy, DEBUG_TYPE, PASS_NAME, false, false)
574
577 std::make_unique<AArch64DAGToDAGISel>(TM, TM.getOptLevel())) {}
578
579/// addBitcastHints - This method adds bitcast hints to the operands of a node
580/// to help instruction selector determine which operands are in Neon registers.
582 SDLoc DL(&N);
583 auto getFloatVT = [&](EVT VT) {
584 EVT ScalarVT = VT.getScalarType();
585 assert((ScalarVT == MVT::i32 || ScalarVT == MVT::i64) && "Unexpected VT");
586 return VT.changeElementType(*(DAG.getContext()),
587 ScalarVT == MVT::i32 ? MVT::f32 : MVT::f64);
588 };
590 NewOps.reserve(N.getNumOperands());
591
592 for (unsigned I = 0, E = N.getNumOperands(); I < E; ++I) {
593 auto bitcasted = DAG.getBitcast(getFloatVT(N.getOperand(I).getValueType()),
594 N.getOperand(I));
595 NewOps.push_back(bitcasted);
596 }
597 EVT OrigVT = N.getValueType(0);
598 SDValue OpNode = DAG.getNode(N.getOpcode(), DL, getFloatVT(OrigVT), NewOps);
599 return DAG.getBitcast(OrigVT, OpNode);
600}
601
602/// isIntImmediate - This method tests to see if the node is a constant
603/// operand. If so Imm will receive the 64-bit value.
604static bool isIntImmediate(const SDNode *N, uint64_t &Imm) {
606 Imm = C->getZExtValue();
607 return true;
608 }
609 return false;
610}
611
612// isIntImmediate - This method tests to see if a constant operand.
613// If so Imm will receive the value.
615 return isIntImmediate(N.getNode(), Imm);
616}
617
618// isOpcWithIntImmediate - This method tests to see if the node is a specific
619// opcode and that it has a immediate integer right operand.
620// If so Imm will receive the 32 bit value.
621static bool isOpcWithIntImmediate(const SDNode *N, unsigned Opc,
622 uint64_t &Imm) {
623 return N->getOpcode() == Opc &&
624 isIntImmediate(N->getOperand(1).getNode(), Imm);
625}
626
627// isIntImmediateEq - This method tests to see if N is a constant operand that
628// is equivalent to 'ImmExpected'.
629#ifndef NDEBUG
630static bool isIntImmediateEq(SDValue N, const uint64_t ImmExpected) {
632 if (!isIntImmediate(N.getNode(), Imm))
633 return false;
634 return Imm == ImmExpected;
635}
636#endif
637
638static APInt DecodeFMOVImm(uint64_t Imm, unsigned RegWidth) {
639 assert(RegWidth == 32 || RegWidth == 64);
640 if (RegWidth == 32)
641 return APInt(RegWidth,
644}
645
646// Decodes the raw integer splat value from a NEON splat operation.
647static std::optional<APInt> DecodeNEONSplat(SDValue N,
648 const AArch64Subtarget *Subtarget) {
649 assert(N.getValueType().isInteger() && "Only integers are supported");
650 if (N->getOpcode() == AArch64ISD::NVCAST ||
651 (N->getOpcode() == ISD::BITCAST && Subtarget->isLittleEndian()))
652 N = N->getOperand(0);
653 unsigned SplatWidth = N.getScalarValueSizeInBits();
654 if (N.getOpcode() == AArch64ISD::FMOV)
655 return DecodeFMOVImm(N.getConstantOperandVal(0), SplatWidth);
656 if (N->getOpcode() == AArch64ISD::MOVI)
657 return APInt(SplatWidth, N.getConstantOperandVal(0));
658 if (N->getOpcode() == AArch64ISD::MOVIshift)
659 return APInt(SplatWidth, N.getConstantOperandVal(0)
660 << N.getConstantOperandVal(1));
661 if (N->getOpcode() == AArch64ISD::MVNIshift)
662 return ~APInt(SplatWidth, N.getConstantOperandVal(0)
663 << N.getConstantOperandVal(1));
664 if (N->getOpcode() == AArch64ISD::MOVIedit)
666 N.getConstantOperandVal(0)));
667 if (N->getOpcode() == AArch64ISD::DUP)
668 if (auto *Const = dyn_cast<ConstantSDNode>(N->getOperand(0)))
669 return Const->getAPIntValue().trunc(SplatWidth);
670 APInt SplatVal;
671 if (ISD::isConstantSplatVector(N.getNode(), SplatVal))
672 return SplatVal.trunc(SplatWidth);
673 // TODO: Recognize more splat-like NEON operations. See ConstantBuildVector
674 // in AArch64ISelLowering.
675 return std::nullopt;
676}
677
678// If \p N is a NEON splat operation (movi, fmov, etc), return the splat value
679// matching the element size of N.
680static std::optional<APInt>
682 unsigned SplatWidth = N.getScalarValueSizeInBits();
683 if (std::optional<APInt> SplatVal = DecodeNEONSplat(N, Subtarget)) {
684 if (SplatVal->getBitWidth() <= SplatWidth)
685 return APInt::getSplat(SplatWidth, *SplatVal);
686 if (SplatVal->isSplat(SplatWidth))
687 return SplatVal->trunc(SplatWidth);
688 }
689 return std::nullopt;
690}
691
692bool AArch64DAGToDAGISel::SelectNEONSplatOfSVELogicalImm(SDValue N,
693 SDValue &Imm) {
694 std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget);
695 if (!ImmVal)
696 return false;
697 uint64_t Encoding;
698 if (!AArch64_AM::isSVELogicalImm(N.getScalarValueSizeInBits(),
699 ImmVal->getZExtValue(), Encoding))
700 return false;
701
702 Imm = CurDAG->getTargetConstant(Encoding, SDLoc(N), MVT::i64);
703 return true;
704}
705
706bool AArch64DAGToDAGISel::SelectNEONSplatOfSVEAddSubImm(SDValue N, SDValue &Imm,
707 SDValue &Shift) {
708 if (std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget))
709 return SelectSVEAddSubImm(SDLoc(N), *ImmVal,
710 N.getValueType().getScalarType().getSimpleVT(),
711 Imm, Shift,
712 /*Negate=*/false);
713 return false;
714}
715
716bool AArch64DAGToDAGISel::SelectNEONSplatOfSVEArithSImm(SDValue N,
717 SDValue &Imm) {
718 if (std::optional<APInt> ImmVal = GetNEONSplatValue(N, Subtarget))
719 return SelectSVESignedArithImm(SDLoc(N), *ImmVal, Imm);
720 return false;
721}
722
723bool AArch64DAGToDAGISel::SelectNEONSplatOfSImm8(SDValue N, SDValue &Imm) {
724 std::optional<APInt> ImmAPIntVal = GetNEONSplatValue(N, Subtarget);
725 if (!ImmAPIntVal)
726 return false;
727
728 int64_t ImmVal = ImmAPIntVal->getSExtValue();
729 if (ImmVal < -128 || ImmVal > 127)
730 return false;
731
732 Imm = CurDAG->getSignedTargetConstant(ImmVal, SDLoc(N), MVT::i32);
733 return true;
734}
735
736bool AArch64DAGToDAGISel::SelectNEONSplatOfUImm8(SDValue N, SDValue &Imm) {
737 std::optional<APInt> ImmAPIntVal = GetNEONSplatValue(N, Subtarget);
738 if (!ImmAPIntVal)
739 return false;
740
741 uint64_t ImmVal = ImmAPIntVal->getZExtValue();
742 if (ImmVal > 255)
743 return false;
744
745 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
746 return true;
747}
748
749bool AArch64DAGToDAGISel::SelectInlineAsmMemoryOperand(
750 const SDValue &Op, const InlineAsm::ConstraintCode ConstraintID,
751 std::vector<SDValue> &OutOps) {
752 switch(ConstraintID) {
753 default:
754 llvm_unreachable("Unexpected asm memory constraint");
755 case InlineAsm::ConstraintCode::m:
756 case InlineAsm::ConstraintCode::o:
757 case InlineAsm::ConstraintCode::Q:
758 // We need to make sure that this one operand does not end up in XZR, thus
759 // require the address to be in a pointer register.
760 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
761 const TargetRegisterClass *TRC =
763 SDLoc dl(Op);
764 SDValue RC = CurDAG->getTargetConstant(TRC->getID(), dl, MVT::i64);
765 SDValue NewOp =
766 SDValue(CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS,
767 dl, Op.getValueType(),
768 Op, RC), 0);
769 OutOps.push_back(NewOp);
770 return false;
771 }
772 return true;
773}
774
775template <unsigned ShiftWidth>
776bool AArch64DAGToDAGISel::SelectShiftMask(SDValue N, SDValue &ShAmt) {
777 // AArch64 shift instructions only use the low log2(ShiftWidth) bits of the
778 // shift amount. If the shift amount has a redundant AND mask that covers
779 // those bits, we can remove it. Return false if nothing was combined so
780 // other patterns (e.g. zext/sext GPR32 → SUBREG_TO_REG) can match.
781 if (N.getOpcode() == ISD::AND && isa<ConstantSDNode>(N.getOperand(1)) &&
782 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
783 uint64_t Mask = N.getConstantOperandVal(1);
784 // Remove AND if the mask covers at least the low log2(ShiftWidth) bits.
785 if ((unsigned)llvm::countr_one(Mask) >= Log2_32(ShiftWidth)) {
786 ShAmt = N.getOperand(0);
787 return true;
788 }
789 }
790 // If shifting by X+/-N where N == 0 mod ShiftWidth, then just shift by X
791 // to avoid the ADD/SUB. The low log2(ShiftWidth) bits are unchanged, so the
792 // shift can use X directly; the original ADD/SUB stays for any other users.
793 if ((N.getOpcode() == ISD::ADD || N.getOpcode() == ISD::SUB) &&
794 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
796 if (isIntImmediate(N.getOperand(1).getNode(), Imm) &&
797 (Imm % ShiftWidth == 0)) {
798 ShAmt = N.getOperand(0);
799 return true;
800 }
801 }
802
803 // If shifting by N-X where N == 0 mod ShiftWidth, then just shift by -X
804 // to generate a NEG instead of a SUB from a constant.
805 if (N.getOpcode() == ISD::SUB && N.hasOneUse() &&
806 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
808 if (isIntImmediate(N.getOperand(0).getNode(), Imm) && Imm != 0 &&
809 (Imm % ShiftWidth == 0)) {
810 SDLoc DL(N);
811 EVT VT = N.getValueType();
812 unsigned NegOpc = (ShiftWidth == 32) ? AArch64::SUBWrr : AArch64::SUBXrr;
813 unsigned ZeroReg = (ShiftWidth == 32) ? AArch64::WZR : AArch64::XZR;
814 SDValue Zero =
815 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, VT);
816 MachineSDNode *Neg =
817 CurDAG->getMachineNode(NegOpc, DL, VT, Zero, N.getOperand(1));
818 ShAmt = SDValue(Neg, 0);
819 return true;
820 }
821 }
822
823 // If shifting by N-X where N == -1 mod ShiftWidth, then just shift by ~X
824 // to generate a NOT (MVN) instead of a SUB from a constant.
825 if (N.getOpcode() == ISD::SUB && N.hasOneUse() &&
826 N.getValueType() == (ShiftWidth == 32 ? MVT::i32 : MVT::i64)) {
828 if (isIntImmediate(N.getOperand(0).getNode(), Imm) &&
829 (Imm % ShiftWidth == ShiftWidth - 1)) {
830 SDLoc DL(N);
831 EVT VT = N.getValueType();
832 unsigned NotOpc = (ShiftWidth == 32) ? AArch64::ORNWrr : AArch64::ORNXrr;
833 unsigned ZeroReg = (ShiftWidth == 32) ? AArch64::WZR : AArch64::XZR;
834 SDValue Zero =
835 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, VT);
836 MachineSDNode *Not =
837 CurDAG->getMachineNode(NotOpc, DL, VT, Zero, N.getOperand(1));
838 ShAmt = SDValue(Not, 0);
839 return true;
840 }
841 }
842
843 return false;
844}
845
846/// SelectArithImmed - Select an immediate value that can be represented as
847/// a 12-bit value shifted left by either 0 or 12. If so, return true with
848/// Val set to the 12-bit value and Shift set to the shifter operand.
849bool AArch64DAGToDAGISel::SelectArithImmed(SDValue N, SDValue &Val,
850 SDValue &Shift) {
851 // This function is called from the addsub_shifted_imm ComplexPattern,
852 // which lists [imm] as the list of opcode it's interested in, however
853 // we still need to check whether the operand is actually an immediate
854 // here because the ComplexPattern opcode list is only used in
855 // root-level opcode matching.
856 if (!isa<ConstantSDNode>(N.getNode()))
857 return false;
858
859 uint64_t Immed = N.getNode()->getAsZExtVal();
860
862 return false;
863
864 unsigned ShiftAmt = AArch64_AM::getArithImmedShift(Immed);
865 Immed >>= ShiftAmt;
866
867 unsigned ShVal = AArch64_AM::getShifterImm(AArch64_AM::LSL, ShiftAmt);
868 SDLoc dl(N);
869 Val = CurDAG->getTargetConstant(Immed, dl, MVT::i32);
870 Shift = CurDAG->getTargetConstant(ShVal, dl, MVT::i32);
871 return true;
872}
873
874/// SelectNegArithImmed - As above, but negates the value before trying to
875/// select it.
876bool AArch64DAGToDAGISel::SelectNegArithImmed(SDValue N, SDValue &Val,
877 SDValue &Shift) {
878 // This function is called from the addsub_shifted_imm ComplexPattern,
879 // which lists [imm] as the list of opcode it's interested in, however
880 // we still need to check whether the operand is actually an immediate
881 // here because the ComplexPattern opcode list is only used in
882 // root-level opcode matching.
883 if (!isa<ConstantSDNode>(N.getNode()))
884 return false;
885
886 // The immediate operand must be a 24-bit zero-extended immediate.
887 uint64_t Immed = N.getNode()->getAsZExtVal();
888
889 // This negation is almost always valid, but "cmp wN, #0" and "cmn wN, #0"
890 // have the opposite effect on the C flag, so this pattern mustn't match under
891 // those circumstances.
892 if (Immed == 0)
893 return false;
894
895 if (N.getValueType() == MVT::i32)
896 Immed = ~((uint32_t)Immed) + 1;
897 else
898 Immed = ~Immed + 1ULL;
899 if (Immed & 0xFFFFFFFFFF000000ULL)
900 return false;
901
902 Immed &= 0xFFFFFFULL;
903 return SelectArithImmed(CurDAG->getConstant(Immed, SDLoc(N), MVT::i32), Val,
904 Shift);
905}
906
907/// getShiftTypeForNode - Translate a shift node to the corresponding
908/// ShiftType value.
910 switch (N.getOpcode()) {
911 default:
913 case ISD::SHL:
914 return AArch64_AM::LSL;
915 case ISD::SRL:
916 return AArch64_AM::LSR;
917 case ISD::SRA:
918 return AArch64_AM::ASR;
919 case ISD::ROTR:
920 return AArch64_AM::ROR;
921 }
922}
923
925 return isa<MemSDNode>(*N) || N->getOpcode() == AArch64ISD::PREFETCH;
926}
927
928/// Determine whether it is worth it to fold SHL into the addressing
929/// mode.
931 assert(V.getOpcode() == ISD::SHL && "invalid opcode");
932 // It is worth folding logical shift of up to three places.
933 auto *CSD = dyn_cast<ConstantSDNode>(V.getOperand(1));
934 if (!CSD)
935 return false;
936 unsigned ShiftVal = CSD->getZExtValue();
937 if (ShiftVal > 3)
938 return false;
939
940 // Check if this particular node is reused in any non-memory related
941 // operation. If yes, do not try to fold this node into the address
942 // computation, since the computation will be kept.
943 const SDNode *Node = V.getNode();
944 for (SDNode *UI : Node->users())
945 if (!isMemOpOrPrefetch(UI))
946 for (SDNode *UII : UI->users())
947 if (!isMemOpOrPrefetch(UII))
948 return false;
949 return true;
950}
951
952/// Determine whether it is worth to fold V into an extended register addressing
953/// mode.
954bool AArch64DAGToDAGISel::isWorthFoldingAddr(SDValue V, unsigned Size) const {
955 // Trivial if we are optimizing for code size or if there is only
956 // one use of the value.
957 if (CurDAG->shouldOptForSize() || V.hasOneUse())
958 return true;
959
960 // If a subtarget has a slow shift, folding a shift into multiple loads
961 // costs additional micro-ops.
962 if (Subtarget->hasAddrLSLSlow14() && (Size == 2 || Size == 16))
963 return false;
964
965 // Check whether we're going to emit the address arithmetic anyway because
966 // it's used by a non-address operation.
967 if (V.getOpcode() == ISD::SHL && isWorthFoldingSHL(V))
968 return true;
969 if (V.getOpcode() == ISD::ADD) {
970 const SDValue LHS = V.getOperand(0);
971 const SDValue RHS = V.getOperand(1);
972 if (LHS.getOpcode() == ISD::SHL && isWorthFoldingSHL(LHS))
973 return true;
974 if (RHS.getOpcode() == ISD::SHL && isWorthFoldingSHL(RHS))
975 return true;
976 }
977
978 // It hurts otherwise, since the value will be reused.
979 return false;
980}
981
982/// and (shl/srl/sra, x, c), mask --> shl (srl/sra, x, c1), c2
983/// to select more shifted register
984bool AArch64DAGToDAGISel::SelectShiftedRegisterFromAnd(SDValue N, SDValue &Reg,
985 SDValue &Shift) {
986 EVT VT = N.getValueType();
987 if (VT != MVT::i32 && VT != MVT::i64)
988 return false;
989
990 if (N->getOpcode() != ISD::AND || !N->hasOneUse())
991 return false;
992 SDValue LHS = N.getOperand(0);
993 if (!LHS->hasOneUse())
994 return false;
995
996 unsigned LHSOpcode = LHS->getOpcode();
997 if (LHSOpcode != ISD::SHL && LHSOpcode != ISD::SRL && LHSOpcode != ISD::SRA)
998 return false;
999
1000 ConstantSDNode *ShiftAmtNode = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
1001 if (!ShiftAmtNode)
1002 return false;
1003
1004 uint64_t ShiftAmtC = ShiftAmtNode->getZExtValue();
1005 ConstantSDNode *RHSC = dyn_cast<ConstantSDNode>(N.getOperand(1));
1006 if (!RHSC)
1007 return false;
1008
1009 APInt AndMask = RHSC->getAPIntValue();
1010 unsigned LowZBits, MaskLen;
1011 if (!AndMask.isShiftedMask(LowZBits, MaskLen))
1012 return false;
1013
1014 unsigned BitWidth = N.getValueSizeInBits();
1015 SDLoc DL(LHS);
1016 uint64_t NewShiftC;
1017 unsigned NewShiftOp;
1018 if (LHSOpcode == ISD::SHL) {
1019 // LowZBits <= ShiftAmtC will fall into isBitfieldPositioningOp
1020 // BitWidth != LowZBits + MaskLen doesn't match the pattern
1021 if (LowZBits <= ShiftAmtC || (BitWidth != LowZBits + MaskLen))
1022 return false;
1023
1024 NewShiftC = LowZBits - ShiftAmtC;
1025 NewShiftOp = VT == MVT::i64 ? AArch64::UBFMXri : AArch64::UBFMWri;
1026 } else {
1027 if (LowZBits == 0)
1028 return false;
1029
1030 // NewShiftC >= BitWidth will fall into isBitfieldExtractOp
1031 NewShiftC = LowZBits + ShiftAmtC;
1032 if (NewShiftC >= BitWidth)
1033 return false;
1034
1035 // SRA need all high bits
1036 if (LHSOpcode == ISD::SRA && (BitWidth != (LowZBits + MaskLen)))
1037 return false;
1038
1039 // SRL high bits can be 0 or 1
1040 if (LHSOpcode == ISD::SRL && (BitWidth > (NewShiftC + MaskLen)))
1041 return false;
1042
1043 if (LHSOpcode == ISD::SRL)
1044 NewShiftOp = VT == MVT::i64 ? AArch64::UBFMXri : AArch64::UBFMWri;
1045 else
1046 NewShiftOp = VT == MVT::i64 ? AArch64::SBFMXri : AArch64::SBFMWri;
1047 }
1048
1049 assert(NewShiftC < BitWidth && "Invalid shift amount");
1050 SDValue NewShiftAmt = CurDAG->getTargetConstant(NewShiftC, DL, VT);
1051 SDValue BitWidthMinus1 = CurDAG->getTargetConstant(BitWidth - 1, DL, VT);
1052 Reg = SDValue(CurDAG->getMachineNode(NewShiftOp, DL, VT, LHS->getOperand(0),
1053 NewShiftAmt, BitWidthMinus1),
1054 0);
1055 unsigned ShVal = AArch64_AM::getShifterImm(AArch64_AM::LSL, LowZBits);
1056 Shift = CurDAG->getTargetConstant(ShVal, DL, MVT::i32);
1057 return true;
1058}
1059
1060/// getExtendTypeForNode - Translate an extend node to the corresponding
1061/// ExtendType value.
1063getExtendTypeForNode(SDValue N, bool IsLoadStore = false) {
1064 if (N.getOpcode() == ISD::SIGN_EXTEND ||
1065 N.getOpcode() == ISD::SIGN_EXTEND_INREG) {
1066 EVT SrcVT;
1067 if (N.getOpcode() == ISD::SIGN_EXTEND_INREG)
1068 SrcVT = cast<VTSDNode>(N.getOperand(1))->getVT();
1069 else
1070 SrcVT = N.getOperand(0).getValueType();
1071
1072 if (!IsLoadStore && SrcVT == MVT::i8)
1073 return AArch64_AM::SXTB;
1074 else if (!IsLoadStore && SrcVT == MVT::i16)
1075 return AArch64_AM::SXTH;
1076 else if (SrcVT == MVT::i32)
1077 return AArch64_AM::SXTW;
1078 assert(SrcVT != MVT::i64 && "extend from 64-bits?");
1079
1081 } else if (N.getOpcode() == ISD::ZERO_EXTEND ||
1082 N.getOpcode() == ISD::ANY_EXTEND) {
1083 EVT SrcVT = N.getOperand(0).getValueType();
1084 if (!IsLoadStore && SrcVT == MVT::i8)
1085 return AArch64_AM::UXTB;
1086 else if (!IsLoadStore && SrcVT == MVT::i16)
1087 return AArch64_AM::UXTH;
1088 else if (SrcVT == MVT::i32)
1089 return AArch64_AM::UXTW;
1090 assert(SrcVT != MVT::i64 && "extend from 64-bits?");
1091
1093 } else if (N.getOpcode() == ISD::AND) {
1094 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1095 if (!CSD)
1097 uint64_t AndMask = CSD->getZExtValue();
1098
1099 switch (AndMask) {
1100 default:
1102 case 0xFF:
1103 return !IsLoadStore ? AArch64_AM::UXTB : AArch64_AM::InvalidShiftExtend;
1104 case 0xFFFF:
1105 return !IsLoadStore ? AArch64_AM::UXTH : AArch64_AM::InvalidShiftExtend;
1106 case 0xFFFFFFFF:
1107 return AArch64_AM::UXTW;
1108 }
1109 }
1110
1112}
1113
1114/// Determine whether constant -V is cheaper to materialise than V.
1115bool AArch64DAGToDAGISel::isWorthNegatingImm(SDValue V) const {
1116 assert(isa<ConstantSDNode>(V) && "invalid node");
1117
1118 EVT VT = V.getValueType();
1119 assert((VT == MVT::i32 || VT == MVT::i64) && "invalid type");
1120
1121 // It's only worth negating the constant if it doesn't have other uses.
1122 if (!V.hasOneUse())
1123 return false;
1124
1125 uint64_t Imm = cast<ConstantSDNode>(V)->getZExtValue();
1126 unsigned BitSize = VT.getSizeInBits();
1128 AArch64_IMM::expandMOVImm(Imm, BitSize, OrigCost);
1129 AArch64_IMM::expandMOVImm(-Imm, BitSize, NewCost);
1130 return NewCost.size() < OrigCost.size();
1131}
1132
1133/// Determine whether it is worth to fold V into an extended register of an
1134/// Add/Sub. LSL means we are folding into an `add w0, w1, w2, lsl #N`
1135/// instruction, and the shift should be treated as worth folding even if has
1136/// multiple uses.
1137bool AArch64DAGToDAGISel::isWorthFoldingALU(SDValue V, bool LSL) const {
1138 // Trivial if we are optimizing for code size or if there is only
1139 // one use of the value.
1140 if (CurDAG->shouldOptForSize() || V.hasOneUse())
1141 return true;
1142
1143 // If a subtarget has a fastpath LSL we can fold a logical shift into
1144 // the add/sub and save a cycle.
1145 if (LSL && Subtarget->hasALULSLFast() && V.getOpcode() == ISD::SHL &&
1146 V.getConstantOperandVal(1) <= 4 &&
1148 return true;
1149
1150 // It hurts otherwise, since the value will be reused.
1151 return false;
1152}
1153
1154/// SelectShiftedRegister - Select a "shifted register" operand. If the value
1155/// is not shifted, set the Shift operand to default of "LSL 0". The logical
1156/// instructions allow the shifted register to be rotated, but the arithmetic
1157/// instructions do not. The AllowROR parameter specifies whether ROR is
1158/// supported.
1159bool AArch64DAGToDAGISel::SelectShiftedRegister(SDValue N, bool AllowROR,
1160 SDValue &Reg, SDValue &Shift) {
1161 if (SelectShiftedRegisterFromAnd(N, Reg, Shift))
1162 return true;
1163
1165 if (ShType == AArch64_AM::InvalidShiftExtend)
1166 return false;
1167 if (!AllowROR && ShType == AArch64_AM::ROR)
1168 return false;
1169
1170 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1171 unsigned BitSize = N.getValueSizeInBits();
1172 unsigned Val = RHS->getZExtValue() & (BitSize - 1);
1173 unsigned ShVal = AArch64_AM::getShifterImm(ShType, Val);
1174
1175 Reg = N.getOperand(0);
1176 Shift = CurDAG->getTargetConstant(ShVal, SDLoc(N), MVT::i32);
1177 return isWorthFoldingALU(N, true);
1178 }
1179
1180 return false;
1181}
1182
1183/// Instructions that accept extend modifiers like UXTW expect the register
1184/// being extended to be a GPR32, but the incoming DAG might be acting on a
1185/// GPR64 (either via SEXT_INREG or AND). Extract the appropriate low bits if
1186/// this is the case.
1188 if (N.getValueType() == MVT::i32)
1189 return N;
1190
1191 SDLoc dl(N);
1192 return CurDAG->getTargetExtractSubreg(AArch64::sub_32, dl, MVT::i32, N);
1193}
1194
1195// Returns a suitable CNT/INC/DEC/RDVL multiplier to calculate VSCALE*N.
1196template<signed Low, signed High, signed Scale>
1197bool AArch64DAGToDAGISel::SelectRDVLImm(SDValue N, SDValue &Imm) {
1198 if (!isa<ConstantSDNode>(N))
1199 return false;
1200
1201 int64_t MulImm = cast<ConstantSDNode>(N)->getSExtValue();
1202 if ((MulImm % std::abs(Scale)) == 0) {
1203 int64_t RDVLImm = MulImm / Scale;
1204 if ((RDVLImm >= Low) && (RDVLImm <= High)) {
1205 Imm = CurDAG->getSignedTargetConstant(RDVLImm, SDLoc(N), MVT::i32);
1206 return true;
1207 }
1208 }
1209
1210 return false;
1211}
1212
1213// Returns a suitable RDSVL multiplier from a left shift.
1214template <signed Low, signed High>
1215bool AArch64DAGToDAGISel::SelectRDSVLShiftImm(SDValue N, SDValue &Imm) {
1216 if (!isa<ConstantSDNode>(N))
1217 return false;
1218
1219 int64_t MulImm = 1LL << cast<ConstantSDNode>(N)->getSExtValue();
1220 if (MulImm >= Low && MulImm <= High) {
1221 Imm = CurDAG->getSignedTargetConstant(MulImm, SDLoc(N), MVT::i32);
1222 return true;
1223 }
1224
1225 return false;
1226}
1227
1228/// SelectArithExtendedRegister - Select a "extended register" operand. This
1229/// operand folds in an extend followed by an optional left shift.
1230bool AArch64DAGToDAGISel::SelectArithExtendedRegister(SDValue N, SDValue &Reg,
1231 SDValue &Shift) {
1232 unsigned ShiftVal = 0;
1234
1235 if (N.getOpcode() == ISD::SHL) {
1236 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1237 if (!CSD)
1238 return false;
1239 ShiftVal = CSD->getZExtValue();
1240 if (ShiftVal > 4)
1241 return false;
1242
1243 Ext = getExtendTypeForNode(N.getOperand(0));
1245 return false;
1246
1247 Reg = N.getOperand(0).getOperand(0);
1248 } else {
1249 Ext = getExtendTypeForNode(N);
1251 return false;
1252
1253 // Don't match sext of vector extracts. These can use SMOV, but if we match
1254 // this as an extended register, we'll always fold the extend into an ALU op
1255 // user of the extend (which results in a UMOV).
1257 SDValue Op = N.getOperand(0);
1258 if (Op->getOpcode() == ISD::ANY_EXTEND)
1259 Op = Op->getOperand(0);
1260 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
1261 Op.getOperand(0).getValueType().isFixedLengthVector())
1262 return false;
1263 }
1264
1265 Reg = N.getOperand(0);
1266
1267 // Don't match if free 32-bit -> 64-bit zext can be used instead. Use the
1268 // isDef32 as a heuristic for when the operand is likely to be a 32bit def.
1269 auto isDef32 = [](SDValue N) {
1270 unsigned Opc = N.getOpcode();
1271 return Opc != ISD::TRUNCATE && Opc != TargetOpcode::EXTRACT_SUBREG &&
1274 Opc != ISD::FREEZE;
1275 };
1276 if (Ext == AArch64_AM::UXTW && Reg->getValueType(0).getSizeInBits() == 32 &&
1277 isDef32(Reg))
1278 return false;
1279 }
1280
1281 // Don't match if the sext can be folded with an asr to form an SBFX.
1282 if (Ext == AArch64_AM::SXTW && Reg.getOpcode() == ISD::SRA &&
1283 Reg.getValueType() == MVT::i32 &&
1284 isa<ConstantSDNode>(Reg.getOperand(1)) && Reg.hasOneUse())
1285 return false;
1286
1287 // AArch64 mandates that the RHS of the operation must use the smallest
1288 // register class that could contain the size being extended from. Thus,
1289 // if we're folding a (sext i8), we need the RHS to be a GPR32, even though
1290 // there might not be an actual 32-bit value in the program. We can
1291 // (harmlessly) synthesize one by injected an EXTRACT_SUBREG here.
1292 assert(Ext != AArch64_AM::UXTX && Ext != AArch64_AM::SXTX);
1293 Reg = narrowIfNeeded(CurDAG, Reg);
1294 Shift = CurDAG->getTargetConstant(getArithExtendImm(Ext, ShiftVal), SDLoc(N),
1295 MVT::i32);
1296 return isWorthFoldingALU(N);
1297}
1298
1299/// SelectArithUXTXRegister - Select a "UXTX register" operand. This
1300/// operand is referred by the instructions have SP operand
1301bool AArch64DAGToDAGISel::SelectArithUXTXRegister(SDValue N, SDValue &Reg,
1302 SDValue &Shift) {
1303 unsigned ShiftVal = 0;
1305
1306 if (N.getOpcode() != ISD::SHL)
1307 return false;
1308
1309 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1310 if (!CSD)
1311 return false;
1312 ShiftVal = CSD->getZExtValue();
1313 if (ShiftVal > 4)
1314 return false;
1315
1316 Ext = AArch64_AM::UXTX;
1317 Reg = N.getOperand(0);
1318 Shift = CurDAG->getTargetConstant(getArithExtendImm(Ext, ShiftVal), SDLoc(N),
1319 MVT::i32);
1320 return isWorthFoldingALU(N);
1321}
1322
1323/// If there's a use of this ADDlow that's not itself a load/store then we'll
1324/// need to create a real ADD instruction from it anyway and there's no point in
1325/// folding it into the mem op. Theoretically, it shouldn't matter, but there's
1326/// a single pseudo-instruction for an ADRP/ADD pair so over-aggressive folding
1327/// leads to duplicated ADRP instructions.
1329 for (auto *User : N->users()) {
1330 if (User->getOpcode() != ISD::LOAD && User->getOpcode() != ISD::STORE &&
1331 User->getOpcode() != ISD::ATOMIC_LOAD &&
1332 User->getOpcode() != ISD::ATOMIC_STORE)
1333 return false;
1334
1335 // ldar and stlr have much more restrictive addressing modes (just a
1336 // register).
1337 if (isStrongerThanMonotonic(cast<MemSDNode>(User)->getSuccessOrdering()))
1338 return false;
1339 }
1340
1341 return true;
1342}
1343
1344/// Check if the immediate offset is valid as a scaled immediate.
1345static bool isValidAsScaledImmediate(int64_t Offset, unsigned Range,
1346 unsigned Size) {
1347 if ((Offset & (Size - 1)) == 0 && Offset >= 0 &&
1348 Offset < (Range << Log2_32(Size)))
1349 return true;
1350 return false;
1351}
1352
1353/// SelectAddrModeIndexedBitWidth - Select a "register plus scaled (un)signed BW-bit
1354/// immediate" address. The "Size" argument is the size in bytes of the memory
1355/// reference, which determines the scale.
1356bool AArch64DAGToDAGISel::SelectAddrModeIndexedBitWidth(SDValue N, bool IsSignedImm,
1357 unsigned BW, unsigned Size,
1358 SDValue &Base,
1359 SDValue &OffImm) {
1360 SDLoc dl(N);
1361 const DataLayout &DL = CurDAG->getDataLayout();
1362 const TargetLowering *TLI = getTargetLowering();
1363 if (N.getOpcode() == ISD::FrameIndex) {
1364 int FI = cast<FrameIndexSDNode>(N)->getIndex();
1365 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1366 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1367 return true;
1368 }
1369
1370 // As opposed to the (12-bit) Indexed addressing mode below, the 7/9-bit signed
1371 // selected here doesn't support labels/immediates, only base+offset.
1372 if (CurDAG->isBaseWithConstantOffset(N)) {
1373 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1374 if (IsSignedImm) {
1375 int64_t RHSC = RHS->getSExtValue();
1376 unsigned Scale = Log2_32(Size);
1377 int64_t Range = 0x1LL << (BW - 1);
1378
1379 if ((RHSC & (Size - 1)) == 0 && RHSC >= -(Range << Scale) &&
1380 RHSC < (Range << Scale)) {
1381 Base = N.getOperand(0);
1382 if (Base.getOpcode() == ISD::FrameIndex) {
1383 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1384 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1385 }
1386 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1387 return true;
1388 }
1389 } else {
1390 // unsigned Immediate
1391 uint64_t RHSC = RHS->getZExtValue();
1392 unsigned Scale = Log2_32(Size);
1393 uint64_t Range = 0x1ULL << BW;
1394
1395 if ((RHSC & (Size - 1)) == 0 && RHSC < (Range << Scale)) {
1396 Base = N.getOperand(0);
1397 if (Base.getOpcode() == ISD::FrameIndex) {
1398 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1399 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1400 }
1401 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1402 return true;
1403 }
1404 }
1405 }
1406 }
1407 // Base only. The address will be materialized into a register before
1408 // the memory is accessed.
1409 // add x0, Xbase, #offset
1410 // stp x1, x2, [x0]
1411 Base = N;
1412 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1413 return true;
1414}
1415
1416/// SelectAddrModeIndexed - Select a "register plus scaled unsigned 12-bit
1417/// immediate" address. The "Size" argument is the size in bytes of the memory
1418/// reference, which determines the scale.
1419bool AArch64DAGToDAGISel::SelectAddrModeIndexed(SDValue N, unsigned Size,
1420 SDValue &Base, SDValue &OffImm) {
1421 SDLoc dl(N);
1422 const DataLayout &DL = CurDAG->getDataLayout();
1423 const TargetLowering *TLI = getTargetLowering();
1424 if (N.getOpcode() == ISD::FrameIndex) {
1425 int FI = cast<FrameIndexSDNode>(N)->getIndex();
1426 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1427 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1428 return true;
1429 }
1430
1431 if (N.getOpcode() == AArch64ISD::ADDlow && isWorthFoldingADDlow(N)) {
1432 GlobalAddressSDNode *GAN =
1433 dyn_cast<GlobalAddressSDNode>(N.getOperand(1).getNode());
1434 Base = N.getOperand(0);
1435 OffImm = N.getOperand(1);
1436 if (!GAN)
1437 return true;
1438
1439 if (GAN->getOffset() % Size == 0 &&
1441 return true;
1442 }
1443
1444 if (CurDAG->isBaseWithConstantOffset(N)) {
1445 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1446 int64_t RHSC = (int64_t)RHS->getZExtValue();
1447 unsigned Scale = Log2_32(Size);
1448 if (isValidAsScaledImmediate(RHSC, 0x1000, Size)) {
1449 Base = N.getOperand(0);
1450 if (Base.getOpcode() == ISD::FrameIndex) {
1451 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1452 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
1453 }
1454 OffImm = CurDAG->getTargetConstant(RHSC >> Scale, dl, MVT::i64);
1455 return true;
1456 }
1457 }
1458 }
1459
1460 // Before falling back to our general case, check if the unscaled
1461 // instructions can handle this. If so, that's preferable.
1462 if (SelectAddrModeUnscaled(N, Size, Base, OffImm))
1463 return false;
1464
1465 // Base only. The address will be materialized into a register before
1466 // the memory is accessed.
1467 // add x0, Xbase, #offset
1468 // ldr x0, [x0]
1469 Base = N;
1470 OffImm = CurDAG->getTargetConstant(0, dl, MVT::i64);
1471 return true;
1472}
1473
1474/// SelectAddrModeUnscaled - Select a "register plus unscaled signed 9-bit
1475/// immediate" address. This should only match when there is an offset that
1476/// is not valid for a scaled immediate addressing mode. The "Size" argument
1477/// is the size in bytes of the memory reference, which is needed here to know
1478/// what is valid for a scaled immediate.
1479bool AArch64DAGToDAGISel::SelectAddrModeUnscaled(SDValue N, unsigned Size,
1480 SDValue &Base,
1481 SDValue &OffImm) {
1482 if (!CurDAG->isBaseWithConstantOffset(N))
1483 return false;
1484 if (ConstantSDNode *RHS = dyn_cast<ConstantSDNode>(N.getOperand(1))) {
1485 int64_t RHSC = RHS->getSExtValue();
1486 if (RHSC >= -256 && RHSC < 256) {
1487 Base = N.getOperand(0);
1488 if (Base.getOpcode() == ISD::FrameIndex) {
1489 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
1490 const TargetLowering *TLI = getTargetLowering();
1491 Base = CurDAG->getTargetFrameIndex(
1492 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
1493 }
1494 OffImm = CurDAG->getTargetConstant(RHSC, SDLoc(N), MVT::i64);
1495 return true;
1496 }
1497 }
1498 return false;
1499}
1500
1502 SDLoc dl(N);
1503 SDValue ImpDef = SDValue(
1504 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, dl, MVT::i64), 0);
1505 return CurDAG->getTargetInsertSubreg(AArch64::sub_32, dl, MVT::i64, ImpDef,
1506 N);
1507}
1508
1509/// Check if the given SHL node (\p N), can be used to form an
1510/// extended register for an addressing mode.
1511bool AArch64DAGToDAGISel::SelectExtendedSHL(SDValue N, unsigned Size,
1512 bool WantExtend, SDValue &Offset,
1513 SDValue &SignExtend) {
1514 assert(N.getOpcode() == ISD::SHL && "Invalid opcode.");
1515 ConstantSDNode *CSD = dyn_cast<ConstantSDNode>(N.getOperand(1));
1516 if (!CSD || (CSD->getZExtValue() & 0x7) != CSD->getZExtValue())
1517 return false;
1518
1519 SDLoc dl(N);
1520 if (WantExtend) {
1522 getExtendTypeForNode(N.getOperand(0), true);
1524 return false;
1525
1526 Offset = narrowIfNeeded(CurDAG, N.getOperand(0).getOperand(0));
1527 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1528 MVT::i32);
1529 } else {
1530 Offset = N.getOperand(0);
1531 SignExtend = CurDAG->getTargetConstant(0, dl, MVT::i32);
1532 }
1533
1534 unsigned LegalShiftVal = Log2_32(Size);
1535 unsigned ShiftVal = CSD->getZExtValue();
1536
1537 if (ShiftVal != 0 && ShiftVal != LegalShiftVal)
1538 return false;
1539
1540 return isWorthFoldingAddr(N, Size);
1541}
1542
1543bool AArch64DAGToDAGISel::SelectAddrModeWRO(SDValue N, unsigned Size,
1544 SDValue &Base, SDValue &Offset,
1545 SDValue &SignExtend,
1546 SDValue &DoShift) {
1547 if (N.getOpcode() != ISD::ADD)
1548 return false;
1549 SDValue LHS = N.getOperand(0);
1550 SDValue RHS = N.getOperand(1);
1551 SDLoc dl(N);
1552
1553 // We don't want to match immediate adds here, because they are better lowered
1554 // to the register-immediate addressing modes.
1556 return false;
1557
1558 // Check if this particular node is reused in any non-memory related
1559 // operation. If yes, do not try to fold this node into the address
1560 // computation, since the computation will be kept.
1561 const SDNode *Node = N.getNode();
1562 for (SDNode *UI : Node->users()) {
1563 if (!isMemOpOrPrefetch(UI))
1564 return false;
1565 }
1566
1567 // Remember if it is worth folding N when it produces extended register.
1568 bool IsExtendedRegisterWorthFolding = isWorthFoldingAddr(N, Size);
1569
1570 // Try to match a shifted extend on the RHS.
1571 if (IsExtendedRegisterWorthFolding && RHS.getOpcode() == ISD::SHL &&
1572 SelectExtendedSHL(RHS, Size, true, Offset, SignExtend)) {
1573 Base = LHS;
1574 DoShift = CurDAG->getTargetConstant(true, dl, MVT::i32);
1575 return true;
1576 }
1577
1578 // Try to match a shifted extend on the LHS.
1579 if (IsExtendedRegisterWorthFolding && LHS.getOpcode() == ISD::SHL &&
1580 SelectExtendedSHL(LHS, Size, true, Offset, SignExtend)) {
1581 Base = RHS;
1582 DoShift = CurDAG->getTargetConstant(true, dl, MVT::i32);
1583 return true;
1584 }
1585
1586 // There was no shift, whatever else we find.
1587 DoShift = CurDAG->getTargetConstant(false, dl, MVT::i32);
1588
1590 // Try to match an unshifted extend on the LHS.
1591 if (IsExtendedRegisterWorthFolding &&
1592 (Ext = getExtendTypeForNode(LHS, true)) !=
1594 Base = RHS;
1595 Offset = narrowIfNeeded(CurDAG, LHS.getOperand(0));
1596 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1597 MVT::i32);
1598 if (isWorthFoldingAddr(LHS, Size))
1599 return true;
1600 }
1601
1602 // Try to match an unshifted extend on the RHS.
1603 if (IsExtendedRegisterWorthFolding &&
1604 (Ext = getExtendTypeForNode(RHS, true)) !=
1606 Base = LHS;
1607 Offset = narrowIfNeeded(CurDAG, RHS.getOperand(0));
1608 SignExtend = CurDAG->getTargetConstant(Ext == AArch64_AM::SXTW, dl,
1609 MVT::i32);
1610 if (isWorthFoldingAddr(RHS, Size))
1611 return true;
1612 }
1613
1614 return false;
1615}
1616
1617// Check if the given immediate is preferred by ADD. If an immediate can be
1618// encoded in an ADD, or it can be encoded in an "ADD LSL #12" and can not be
1619// encoded by one MOVZ, return true.
1620static bool isPreferredADD(int64_t ImmOff) {
1621 // Constant in [0x0, 0xfff] can be encoded in ADD.
1622 if ((ImmOff & 0xfffffffffffff000LL) == 0x0LL)
1623 return true;
1624 // Check if it can be encoded in an "ADD LSL #12".
1625 if ((ImmOff & 0xffffffffff000fffLL) == 0x0LL)
1626 // As a single MOVZ is faster than a "ADD of LSL #12", ignore such constant.
1627 return (ImmOff & 0xffffffffff00ffffLL) != 0x0LL &&
1628 (ImmOff & 0xffffffffffff0fffLL) != 0x0LL;
1629 return false;
1630}
1631
1632bool AArch64DAGToDAGISel::SelectAddrModeXRO(SDValue N, unsigned Size,
1633 SDValue &Base, SDValue &Offset,
1634 SDValue &SignExtend,
1635 SDValue &DoShift) {
1636 if (N.getOpcode() != ISD::ADD)
1637 return false;
1638 SDValue LHS = N.getOperand(0);
1639 SDValue RHS = N.getOperand(1);
1640 SDLoc DL(N);
1641
1642 // Check if this particular node is reused in any non-memory related
1643 // operation. If yes, do not try to fold this node into the address
1644 // computation, since the computation will be kept.
1645 const SDNode *Node = N.getNode();
1646 for (SDNode *UI : Node->users()) {
1647 if (!isMemOpOrPrefetch(UI))
1648 return false;
1649 }
1650
1651 // Watch out if RHS is a wide immediate, it can not be selected into
1652 // [BaseReg+Imm] addressing mode. Also it may not be able to be encoded into
1653 // ADD/SUB. Instead it will use [BaseReg + 0] address mode and generate
1654 // instructions like:
1655 // MOV X0, WideImmediate
1656 // ADD X1, BaseReg, X0
1657 // LDR X2, [X1, 0]
1658 // For such situation, using [BaseReg, XReg] addressing mode can save one
1659 // ADD/SUB:
1660 // MOV X0, WideImmediate
1661 // LDR X2, [BaseReg, X0]
1662 if (isa<ConstantSDNode>(RHS)) {
1663 int64_t ImmOff = (int64_t)RHS->getAsZExtVal();
1664 // Skip the immediate can be selected by load/store addressing mode.
1665 // Also skip the immediate can be encoded by a single ADD (SUB is also
1666 // checked by using -ImmOff).
1667 if (isValidAsScaledImmediate(ImmOff, 0x1000, Size) ||
1668 isPreferredADD(ImmOff) || isPreferredADD(-ImmOff))
1669 return false;
1670
1671 SDValue Ops[] = { RHS };
1672 SDNode *MOVI =
1673 CurDAG->getMachineNode(AArch64::MOVi64imm, DL, MVT::i64, Ops);
1674 SDValue MOVIV = SDValue(MOVI, 0);
1675 // This ADD of two X register will be selected into [Reg+Reg] mode.
1676 N = CurDAG->getNode(ISD::ADD, DL, MVT::i64, LHS, MOVIV);
1677 }
1678
1679 // Remember if it is worth folding N when it produces extended register.
1680 bool IsExtendedRegisterWorthFolding = isWorthFoldingAddr(N, Size);
1681
1682 // Try to match a shifted extend on the RHS.
1683 if (IsExtendedRegisterWorthFolding && RHS.getOpcode() == ISD::SHL &&
1684 SelectExtendedSHL(RHS, Size, false, Offset, SignExtend)) {
1685 Base = LHS;
1686 DoShift = CurDAG->getTargetConstant(true, DL, MVT::i32);
1687 return true;
1688 }
1689
1690 // Try to match a shifted extend on the LHS.
1691 if (IsExtendedRegisterWorthFolding && LHS.getOpcode() == ISD::SHL &&
1692 SelectExtendedSHL(LHS, Size, false, Offset, SignExtend)) {
1693 Base = RHS;
1694 DoShift = CurDAG->getTargetConstant(true, DL, MVT::i32);
1695 return true;
1696 }
1697
1698 // Match any non-shifted, non-extend, non-immediate add expression.
1699 Base = LHS;
1700 Offset = RHS;
1701 SignExtend = CurDAG->getTargetConstant(false, DL, MVT::i32);
1702 DoShift = CurDAG->getTargetConstant(false, DL, MVT::i32);
1703 // Reg1 + Reg2 is free: no check needed.
1704 return true;
1705}
1706
1707SDValue AArch64DAGToDAGISel::createDTuple(ArrayRef<SDValue> Regs) {
1708 static const unsigned RegClassIDs[] = {
1709 AArch64::DDRegClassID, AArch64::DDDRegClassID, AArch64::DDDDRegClassID};
1710 static const unsigned SubRegs[] = {AArch64::dsub0, AArch64::dsub1,
1711 AArch64::dsub2, AArch64::dsub3};
1712
1713 return createTuple(Regs, RegClassIDs, SubRegs);
1714}
1715
1716SDValue AArch64DAGToDAGISel::createQTuple(ArrayRef<SDValue> Regs) {
1717 static const unsigned RegClassIDs[] = {
1718 AArch64::QQRegClassID, AArch64::QQQRegClassID, AArch64::QQQQRegClassID};
1719 static const unsigned SubRegs[] = {AArch64::qsub0, AArch64::qsub1,
1720 AArch64::qsub2, AArch64::qsub3};
1721
1722 return createTuple(Regs, RegClassIDs, SubRegs);
1723}
1724
1725SDValue AArch64DAGToDAGISel::createZTuple(ArrayRef<SDValue> Regs) {
1726 static const unsigned RegClassIDs[] = {AArch64::ZPR2RegClassID,
1727 AArch64::ZPR3RegClassID,
1728 AArch64::ZPR4RegClassID};
1729 static const unsigned SubRegs[] = {AArch64::zsub0, AArch64::zsub1,
1730 AArch64::zsub2, AArch64::zsub3};
1731
1732 return createTuple(Regs, RegClassIDs, SubRegs);
1733}
1734
1735SDValue AArch64DAGToDAGISel::createZMulTuple(ArrayRef<SDValue> Regs) {
1736 assert(Regs.size() == 2 || Regs.size() == 4);
1737
1738 // The createTuple interface requires 3 RegClassIDs for each possible
1739 // tuple type even though we only have them for ZPR2 and ZPR4.
1740 static const unsigned RegClassIDs[] = {AArch64::ZPR2Mul2RegClassID, 0,
1741 AArch64::ZPR4Mul4RegClassID};
1742 static const unsigned SubRegs[] = {AArch64::zsub0, AArch64::zsub1,
1743 AArch64::zsub2, AArch64::zsub3};
1744 return createTuple(Regs, RegClassIDs, SubRegs);
1745}
1746
1747SDValue AArch64DAGToDAGISel::createTuple(ArrayRef<SDValue> Regs,
1748 const unsigned RegClassIDs[],
1749 const unsigned SubRegs[]) {
1750 // There's no special register-class for a vector-list of 1 element: it's just
1751 // a vector.
1752 if (Regs.size() == 1)
1753 return Regs[0];
1754
1755 assert(Regs.size() >= 2 && Regs.size() <= 4);
1756
1757 SDLoc DL(Regs[0]);
1758
1760
1761 // First operand of REG_SEQUENCE is the desired RegClass.
1762 Ops.push_back(
1763 CurDAG->getTargetConstant(RegClassIDs[Regs.size() - 2], DL, MVT::i32));
1764
1765 // Then we get pairs of source & subregister-position for the components.
1766 for (unsigned i = 0; i < Regs.size(); ++i) {
1767 Ops.push_back(Regs[i]);
1768 Ops.push_back(CurDAG->getTargetConstant(SubRegs[i], DL, MVT::i32));
1769 }
1770
1771 SDNode *N =
1772 CurDAG->getMachineNode(TargetOpcode::REG_SEQUENCE, DL, MVT::Untyped, Ops);
1773 return SDValue(N, 0);
1774}
1775
1776void AArch64DAGToDAGISel::SelectTable(SDNode *N, unsigned NumVecs, unsigned Opc,
1777 bool isExt) {
1778 SDLoc dl(N);
1779 EVT VT = N->getValueType(0);
1780
1781 unsigned ExtOff = isExt;
1782
1783 // Form a REG_SEQUENCE to force register allocation.
1784 unsigned Vec0Off = ExtOff + 1;
1785 SmallVector<SDValue, 4> Regs(N->ops().slice(Vec0Off, NumVecs));
1786 SDValue RegSeq = createQTuple(Regs);
1787
1789 if (isExt)
1790 Ops.push_back(N->getOperand(1));
1791 Ops.push_back(RegSeq);
1792 Ops.push_back(N->getOperand(NumVecs + ExtOff + 1));
1793 ReplaceNode(N, CurDAG->getMachineNode(Opc, dl, VT, Ops));
1794}
1795
1796static std::tuple<SDValue, SDValue>
1798 SDLoc DL(Disc);
1799 SDValue AddrDisc;
1800 SDValue ConstDisc;
1801
1802 // If this is a blend, remember the constant and address discriminators.
1803 // Otherwise, it's either a constant discriminator, or a non-blended
1804 // address discriminator.
1805 if (Disc->getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
1806 Disc->getConstantOperandVal(0) == Intrinsic::ptrauth_blend) {
1807 AddrDisc = Disc->getOperand(1);
1808 ConstDisc = Disc->getOperand(2);
1809 } else {
1810 ConstDisc = Disc;
1811 }
1812
1813 // If the constant discriminator (either the blend RHS, or the entire
1814 // discriminator value) isn't a 16-bit constant, bail out, and let the
1815 // discriminator be computed separately.
1816 auto *ConstDiscN = dyn_cast<ConstantSDNode>(ConstDisc);
1817 if (!ConstDiscN || !isUInt<16>(ConstDiscN->getZExtValue()))
1818 return std::make_tuple(DAG->getTargetConstant(0, DL, MVT::i64), Disc);
1819
1820 // If there's no address discriminator, use XZR directly.
1821 if (!AddrDisc)
1822 AddrDisc = DAG->getRegister(AArch64::XZR, MVT::i64);
1823
1824 return std::make_tuple(
1825 DAG->getTargetConstant(ConstDiscN->getZExtValue(), DL, MVT::i64),
1826 AddrDisc);
1827}
1828
1829void AArch64DAGToDAGISel::SelectPtrauthAuth(SDNode *N) {
1830 SDLoc DL(N);
1831 // IntrinsicID is operand #0
1832 SDValue Val = N->getOperand(1);
1833 SDValue AUTKey = N->getOperand(2);
1834 SDValue AUTDisc = N->getOperand(3);
1835
1836 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1837 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1838
1839 SDValue AUTAddrDisc, AUTConstDisc;
1840 std::tie(AUTConstDisc, AUTAddrDisc) =
1841 extractPtrauthBlendDiscriminators(AUTDisc, CurDAG);
1842
1843 if (!Subtarget->isX16X17Safer()) {
1844 std::vector<SDValue> Ops = {Val, AUTKey, AUTConstDisc, AUTAddrDisc};
1845 // Copy deactivation symbol if present.
1846 if (N->getNumOperands() > 4)
1847 Ops.push_back(N->getOperand(4));
1848
1849 SDNode *AUT =
1850 CurDAG->getMachineNode(AArch64::AUTxMxN, DL, MVT::i64, MVT::i64, Ops);
1851 ReplaceNode(N, AUT);
1852 } else {
1853 SDValue X16Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1854 AArch64::X16, Val, SDValue());
1855 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc, X16Copy.getValue(1)};
1856
1857 SDNode *AUT = CurDAG->getMachineNode(AArch64::AUTx16x17, DL, MVT::i64, Ops);
1858 ReplaceNode(N, AUT);
1859 }
1860}
1861
1862void AArch64DAGToDAGISel::SelectPtrauthResign(SDNode *N) {
1863 SDLoc DL(N);
1864 // IntrinsicID is operand #0, if W_CHAIN it is #1
1865 int OffsetBase = N->getOpcode() == ISD::INTRINSIC_W_CHAIN ? 1 : 0;
1866 SDValue Val = N->getOperand(OffsetBase + 1);
1867 SDValue AUTKey = N->getOperand(OffsetBase + 2);
1868 SDValue AUTDisc = N->getOperand(OffsetBase + 3);
1869 SDValue PACKey = N->getOperand(OffsetBase + 4);
1870 SDValue PACDisc = N->getOperand(OffsetBase + 5);
1871 uint32_t IntNum = N->getConstantOperandVal(OffsetBase + 0);
1872 bool HasLoad = IntNum == Intrinsic::ptrauth_resign_load_relative;
1873
1874 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1875 unsigned PACKeyC = cast<ConstantSDNode>(PACKey)->getZExtValue();
1876
1877 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1878 PACKey = CurDAG->getTargetConstant(PACKeyC, DL, MVT::i64);
1879
1880 SDValue AUTAddrDisc, AUTConstDisc;
1881 std::tie(AUTConstDisc, AUTAddrDisc) =
1882 extractPtrauthBlendDiscriminators(AUTDisc, CurDAG);
1883
1884 SDValue PACAddrDisc, PACConstDisc;
1885 std::tie(PACConstDisc, PACAddrDisc) =
1886 extractPtrauthBlendDiscriminators(PACDisc, CurDAG);
1887
1888 SDValue X16Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1889 AArch64::X16, Val, SDValue());
1890
1891 if (HasLoad) {
1892 SDValue Addend = N->getOperand(OffsetBase + 6);
1893 SDValue IncomingChain = N->getOperand(0);
1894 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc,
1895 PACKey, PACConstDisc, PACAddrDisc,
1896 Addend, IncomingChain, X16Copy.getValue(1)};
1897
1898 SDNode *AUTRELLOADPAC = CurDAG->getMachineNode(AArch64::AUTRELLOADPAC, DL,
1899 MVT::i64, MVT::Other, Ops);
1900 ReplaceNode(N, AUTRELLOADPAC);
1901 } else {
1902 SDValue Ops[] = {AUTKey, AUTConstDisc, AUTAddrDisc, PACKey,
1903 PACConstDisc, PACAddrDisc, X16Copy.getValue(1)};
1904
1905 SDNode *AUTPAC = CurDAG->getMachineNode(AArch64::AUTPAC, DL, MVT::i64, Ops);
1906 ReplaceNode(N, AUTPAC);
1907 }
1908}
1909
1910void AArch64DAGToDAGISel::SelectPtrauthResignWithPC(SDNode *N) {
1911 SDLoc DL(N);
1912 SDValue Val = N->getOperand(1);
1913 SDValue AUTKey = N->getOperand(2);
1914 SDValue AUTDisc = N->getOperand(3);
1915 SDValue AUTPC = N->getOperand(4);
1916 SDValue PACKey = N->getOperand(5);
1917 SDValue PACDisc = N->getOperand(6);
1918
1919 unsigned AUTKeyC = cast<ConstantSDNode>(AUTKey)->getZExtValue();
1920 unsigned PACKeyC = cast<ConstantSDNode>(PACKey)->getZExtValue();
1921
1922 AUTKey = CurDAG->getTargetConstant(AUTKeyC, DL, MVT::i64);
1923 PACKey = CurDAG->getTargetConstant(PACKeyC, DL, MVT::i64);
1924
1925 SDValue PACAddrDisc, PACConstDisc;
1926 std::tie(PACConstDisc, PACAddrDisc) =
1927 extractPtrauthBlendDiscriminators(PACDisc, CurDAG);
1928
1929 SDValue X17Copy = CurDAG->getCopyToReg(CurDAG->getEntryNode(), DL,
1930 AArch64::X17, Val, SDValue());
1931 SDValue X16Copy = CurDAG->getCopyToReg(
1932 CurDAG->getEntryNode(), DL, AArch64::X16, AUTDisc, X17Copy.getValue(1));
1933 SDValue X15Copy = CurDAG->getCopyToReg(
1934 CurDAG->getEntryNode(), DL, AArch64::X15, AUTPC, X16Copy.getValue(1));
1935
1936 SDValue Ops[] = {AUTKey, PACKey, PACConstDisc, PACAddrDisc,
1937 X15Copy.getValue(1)};
1938 SDNode *AUTPCPAC =
1939 CurDAG->getMachineNode(AArch64::AUTPCPAC, DL, MVT::i64, Ops);
1940 ReplaceNode(N, AUTPCPAC);
1941}
1942
1943bool AArch64DAGToDAGISel::tryIndexedLoad(SDNode *N) {
1944 LoadSDNode *LD = cast<LoadSDNode>(N);
1945 if (LD->isUnindexed())
1946 return false;
1947 EVT VT = LD->getMemoryVT();
1948 EVT DstVT = N->getValueType(0);
1949 ISD::MemIndexedMode AM = LD->getAddressingMode();
1950 bool IsPre = AM == ISD::PRE_INC || AM == ISD::PRE_DEC;
1951 ConstantSDNode *OffsetOp = cast<ConstantSDNode>(LD->getOffset());
1952 int OffsetVal = (int)OffsetOp->getZExtValue();
1953
1954 // We're not doing validity checking here. That was done when checking
1955 // if we should mark the load as indexed or not. We're just selecting
1956 // the right instruction.
1957 unsigned Opcode = 0;
1958
1959 ISD::LoadExtType ExtType = LD->getExtensionType();
1960 bool InsertTo64 = false;
1961 bool UseLd1 =
1962 (VT.is64BitVector() || VT.is128BitVector()) &&
1963 (!Subtarget->isLittleEndian() || (Subtarget->requiresStrictAlign() &&
1964 LD->getAlign() < VT.getStoreSize()));
1965 if (VT == MVT::i64)
1966 Opcode = IsPre ? AArch64::LDRXpre : AArch64::LDRXpost;
1967 else if (VT == MVT::i32) {
1968 if (ExtType == ISD::NON_EXTLOAD)
1969 Opcode = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
1970 else if (ExtType == ISD::SEXTLOAD)
1971 Opcode = IsPre ? AArch64::LDRSWpre : AArch64::LDRSWpost;
1972 else {
1973 Opcode = IsPre ? AArch64::LDRWpre : AArch64::LDRWpost;
1974 InsertTo64 = true;
1975 // The result of the load is only i32. It's the subreg_to_reg that makes
1976 // it into an i64.
1977 DstVT = MVT::i32;
1978 }
1979 } else if (VT == MVT::i16) {
1980 if (ExtType == ISD::SEXTLOAD) {
1981 if (DstVT == MVT::i64)
1982 Opcode = IsPre ? AArch64::LDRSHXpre : AArch64::LDRSHXpost;
1983 else
1984 Opcode = IsPre ? AArch64::LDRSHWpre : AArch64::LDRSHWpost;
1985 } else {
1986 Opcode = IsPre ? AArch64::LDRHHpre : AArch64::LDRHHpost;
1987 InsertTo64 = DstVT == MVT::i64;
1988 // The result of the load is only i32. It's the subreg_to_reg that makes
1989 // it into an i64.
1990 DstVT = MVT::i32;
1991 }
1992 } else if (VT == MVT::i8) {
1993 if (ExtType == ISD::SEXTLOAD) {
1994 if (DstVT == MVT::i64)
1995 Opcode = IsPre ? AArch64::LDRSBXpre : AArch64::LDRSBXpost;
1996 else
1997 Opcode = IsPre ? AArch64::LDRSBWpre : AArch64::LDRSBWpost;
1998 } else {
1999 Opcode = IsPre ? AArch64::LDRBBpre : AArch64::LDRBBpost;
2000 InsertTo64 = DstVT == MVT::i64;
2001 // The result of the load is only i32. It's the subreg_to_reg that makes
2002 // it into an i64.
2003 DstVT = MVT::i32;
2004 }
2005 } else if (VT == MVT::f16) {
2006 Opcode = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
2007 } else if (VT == MVT::bf16) {
2008 Opcode = IsPre ? AArch64::LDRHpre : AArch64::LDRHpost;
2009 } else if (VT == MVT::f32) {
2010 Opcode = IsPre ? AArch64::LDRSpre : AArch64::LDRSpost;
2011 } else if (VT == MVT::f64 || (VT.is64BitVector() && !UseLd1)) {
2012 Opcode = IsPre ? AArch64::LDRDpre : AArch64::LDRDpost;
2013 } else if (VT.is128BitVector() && !UseLd1) {
2014 Opcode = IsPre ? AArch64::LDRQpre : AArch64::LDRQpost;
2015 } else if (VT.is64BitVector() && UseLd1) {
2016 if (IsPre || OffsetVal != 8)
2017 return false;
2018 switch (VT.getScalarSizeInBits()) {
2019 case 8:
2020 Opcode = AArch64::LD1Onev8b_POST;
2021 break;
2022 case 16:
2023 Opcode = AArch64::LD1Onev4h_POST;
2024 break;
2025 case 32:
2026 Opcode = AArch64::LD1Onev2s_POST;
2027 break;
2028 case 64:
2029 Opcode = AArch64::LD1Onev1d_POST;
2030 break;
2031 default:
2032 llvm_unreachable("Expected vector element to be a power of 2");
2033 }
2034 } else if (VT.is128BitVector() && UseLd1) {
2035 if (IsPre || OffsetVal != 16)
2036 return false;
2037 switch (VT.getScalarSizeInBits()) {
2038 case 8:
2039 Opcode = AArch64::LD1Onev16b_POST;
2040 break;
2041 case 16:
2042 Opcode = AArch64::LD1Onev8h_POST;
2043 break;
2044 case 32:
2045 Opcode = AArch64::LD1Onev4s_POST;
2046 break;
2047 case 64:
2048 Opcode = AArch64::LD1Onev2d_POST;
2049 break;
2050 default:
2051 llvm_unreachable("Expected vector element to be a power of 2");
2052 }
2053 } else
2054 return false;
2055 SDValue Chain = LD->getChain();
2056 SDValue Base = LD->getBasePtr();
2057 SDLoc dl(N);
2058 // LD1 encodes an immediate offset by using XZR as the offset register.
2059 SDValue Offset = UseLd1 ? CurDAG->getRegister(AArch64::XZR, MVT::i64)
2060 : CurDAG->getTargetConstant(OffsetVal, dl, MVT::i64);
2061 SDValue Ops[] = { Base, Offset, Chain };
2062 SDNode *Res = CurDAG->getMachineNode(Opcode, dl, MVT::i64, DstVT,
2063 MVT::Other, Ops);
2064
2065 // Transfer memoperands.
2066 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
2067 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Res), {MemOp});
2068
2069 // Either way, we're replacing the node, so tell the caller that.
2070 SDValue LoadedVal = SDValue(Res, 1);
2071 if (InsertTo64) {
2072 SDValue SubReg = CurDAG->getTargetConstant(AArch64::sub_32, dl, MVT::i32);
2073 LoadedVal = SDValue(CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, dl,
2074 MVT::i64, LoadedVal, SubReg),
2075 0);
2076 }
2077
2078 ReplaceUses(SDValue(N, 0), LoadedVal);
2079 ReplaceUses(SDValue(N, 1), SDValue(Res, 0));
2080 ReplaceUses(SDValue(N, 2), SDValue(Res, 2));
2081 CurDAG->RemoveDeadNode(N);
2082 return true;
2083}
2084
2085void AArch64DAGToDAGISel::SelectLoad(SDNode *N, unsigned NumVecs, unsigned Opc,
2086 unsigned SubRegIdx) {
2087 SDLoc dl(N);
2088 EVT VT = N->getValueType(0);
2089 SDValue Chain = N->getOperand(0);
2090
2091 SDValue Ops[] = {N->getOperand(2), // Mem operand;
2092 Chain};
2093
2094 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2095
2096 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2097 SDValue SuperReg = SDValue(Ld, 0);
2098 for (unsigned i = 0; i < NumVecs; ++i)
2099 ReplaceUses(SDValue(N, i),
2100 CurDAG->getTargetExtractSubreg(SubRegIdx + i, dl, VT, SuperReg));
2101
2102 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 1));
2103
2104 // Transfer memoperands. In the case of AArch64::LD64B, there won't be one,
2105 // because it's too simple to have needed special treatment during lowering.
2106 if (auto *MemIntr = dyn_cast<MemIntrinsicSDNode>(N)) {
2107 MachineMemOperand *MemOp = MemIntr->getMemOperand();
2108 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
2109 }
2110
2111 CurDAG->RemoveDeadNode(N);
2112}
2113
2114void AArch64DAGToDAGISel::SelectPostLoad(SDNode *N, unsigned NumVecs,
2115 unsigned Opc, unsigned SubRegIdx) {
2116 SDLoc dl(N);
2117 EVT VT = N->getValueType(0);
2118 SDValue Chain = N->getOperand(0);
2119
2120 SDValue Ops[] = {N->getOperand(1), // Mem operand
2121 N->getOperand(2), // Incremental
2122 Chain};
2123
2124 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2125 MVT::Untyped, MVT::Other};
2126
2127 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2128
2129 // Update uses of write back register
2130 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 0));
2131
2132 // Update uses of vector list
2133 SDValue SuperReg = SDValue(Ld, 1);
2134 if (NumVecs == 1)
2135 ReplaceUses(SDValue(N, 0), SuperReg);
2136 else
2137 for (unsigned i = 0; i < NumVecs; ++i)
2138 ReplaceUses(SDValue(N, i),
2139 CurDAG->getTargetExtractSubreg(SubRegIdx + i, dl, VT, SuperReg));
2140
2141 // Transfer memoperands.
2142 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2143 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
2144
2145 // Update the chain
2146 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(Ld, 2));
2147 CurDAG->RemoveDeadNode(N);
2148}
2149
2150/// Optimize \param OldBase and \param OldOffset selecting the best addressing
2151/// mode. Returns a tuple consisting of an Opcode, an SDValue representing the
2152/// new Base and an SDValue representing the new offset.
2153std::tuple<unsigned, SDValue, SDValue>
2154AArch64DAGToDAGISel::findAddrModeSVELoadStore(SDNode *N, unsigned Opc_rr,
2155 unsigned Opc_ri,
2156 const SDValue &OldBase,
2157 const SDValue &OldOffset,
2158 unsigned Scale) {
2159 SDValue NewBase = OldBase;
2160 SDValue NewOffset = OldOffset;
2161 // Detect a possible Reg+Imm addressing mode.
2162 const bool IsRegImm = SelectAddrModeIndexedSVE</*Min=*/-8, /*Max=*/7>(
2163 N, OldBase, NewBase, NewOffset);
2164
2165 // Detect a possible reg+reg addressing mode, but only if we haven't already
2166 // detected a Reg+Imm one.
2167 const bool IsRegReg =
2168 !IsRegImm && SelectSVERegRegAddrMode(OldBase, Scale, NewBase, NewOffset);
2169
2170 // Select the instruction.
2171 return std::make_tuple(IsRegReg ? Opc_rr : Opc_ri, NewBase, NewOffset);
2172}
2173
2174enum class SelectTypeKind {
2175 Int1 = 0,
2176 Int = 1,
2177 FP = 2,
2179};
2180
2181/// This function selects an opcode from a list of opcodes, which is
2182/// expected to be the opcode for { 8-bit, 16-bit, 32-bit, 64-bit }
2183/// element types, in this order.
2184template <SelectTypeKind Kind>
2185static unsigned SelectOpcodeFromVT(EVT VT, ArrayRef<unsigned> Opcodes) {
2186 // Only match scalable vector VTs
2187 if (!VT.isScalableVector())
2188 return 0;
2189
2190 EVT EltVT = VT.getVectorElementType();
2191 unsigned Key = VT.getVectorMinNumElements();
2192 switch (Kind) {
2194 break;
2196 if (EltVT != MVT::i8 && EltVT != MVT::i16 && EltVT != MVT::i32 &&
2197 EltVT != MVT::i64)
2198 return 0;
2199 break;
2201 if (EltVT != MVT::i1)
2202 return 0;
2203 break;
2204 case SelectTypeKind::FP:
2205 if (EltVT == MVT::bf16)
2206 Key = 16;
2207 else if (EltVT != MVT::bf16 && EltVT != MVT::f16 && EltVT != MVT::f32 &&
2208 EltVT != MVT::f64)
2209 return 0;
2210 break;
2211 }
2212
2213 unsigned Offset;
2214 switch (Key) {
2215 case 16: // 8-bit or bf16
2216 Offset = 0;
2217 break;
2218 case 8: // 16-bit
2219 Offset = 1;
2220 break;
2221 case 4: // 32-bit
2222 Offset = 2;
2223 break;
2224 case 2: // 64-bit
2225 Offset = 3;
2226 break;
2227 default:
2228 return 0;
2229 }
2230
2231 return (Opcodes.size() <= Offset) ? 0 : Opcodes[Offset];
2232}
2233
2234// This function is almost identical to SelectWhilePair, but has an
2235// extra check on the range of the immediate operand.
2236// TODO: Merge these two functions together at some point?
2237void AArch64DAGToDAGISel::SelectPExtPair(SDNode *N, unsigned Opc) {
2238 // Immediate can be either 0 or 1.
2239 if (ConstantSDNode *Imm = dyn_cast<ConstantSDNode>(N->getOperand(2)))
2240 if (Imm->getZExtValue() > 1)
2241 return;
2242
2243 SDLoc DL(N);
2244 EVT VT = N->getValueType(0);
2245 SDValue Ops[] = {N->getOperand(1), N->getOperand(2)};
2246 SDNode *WhilePair = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2247 SDValue SuperReg = SDValue(WhilePair, 0);
2248
2249 for (unsigned I = 0; I < 2; ++I)
2250 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2251 AArch64::psub0 + I, DL, VT, SuperReg));
2252
2253 CurDAG->RemoveDeadNode(N);
2254}
2255
2256void AArch64DAGToDAGISel::SelectWhilePair(SDNode *N, unsigned Opc) {
2257 SDLoc DL(N);
2258 EVT VT = N->getValueType(0);
2259
2260 SDValue Ops[] = {N->getOperand(1), N->getOperand(2)};
2261
2262 SDNode *WhilePair = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2263 SDValue SuperReg = SDValue(WhilePair, 0);
2264
2265 for (unsigned I = 0; I < 2; ++I)
2266 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2267 AArch64::psub0 + I, DL, VT, SuperReg));
2268
2269 CurDAG->RemoveDeadNode(N);
2270}
2271
2272void AArch64DAGToDAGISel::SelectCVTIntrinsic(SDNode *N, unsigned NumVecs,
2273 unsigned Opcode) {
2274 EVT VT = N->getValueType(0);
2275 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2276 SDValue Ops = createZTuple(Regs);
2277 SDLoc DL(N);
2278 SDNode *Intrinsic = CurDAG->getMachineNode(Opcode, DL, MVT::Untyped, Ops);
2279 SDValue SuperReg = SDValue(Intrinsic, 0);
2280 for (unsigned i = 0; i < NumVecs; ++i)
2281 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2282 AArch64::zsub0 + i, DL, VT, SuperReg));
2283
2284 CurDAG->RemoveDeadNode(N);
2285}
2286
2287void AArch64DAGToDAGISel::SelectCVTIntrinsicFP8(SDNode *N, unsigned NumVecs,
2288 unsigned Opcode) {
2289 SDLoc DL(N);
2290 EVT VT = N->getValueType(0);
2291 SmallVector<SDValue, 4> Ops(N->op_begin() + 2, N->op_end());
2292 Ops.push_back(/*Chain*/ N->getOperand(0));
2293
2294 SDNode *Instruction =
2295 CurDAG->getMachineNode(Opcode, DL, {MVT::Untyped, MVT::Other}, Ops);
2296 SDValue SuperReg = SDValue(Instruction, 0);
2297
2298 for (unsigned i = 0; i < NumVecs; ++i)
2299 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2300 AArch64::zsub0 + i, DL, VT, SuperReg));
2301
2302 // Copy chain
2303 unsigned ChainIdx = NumVecs;
2304 ReplaceUses(SDValue(N, ChainIdx), SDValue(Instruction, 1));
2305 CurDAG->RemoveDeadNode(N);
2306}
2307
2308void AArch64DAGToDAGISel::SelectDestructiveMultiIntrinsic(SDNode *N,
2309 unsigned NumVecs,
2310 bool IsZmMulti,
2311 unsigned Opcode,
2312 bool HasPred) {
2313 assert(Opcode != 0 && "Unexpected opcode");
2314
2315 SDLoc DL(N);
2316 EVT VT = N->getValueType(0);
2317 SDUse *OpsIter = N->op_begin() + 1; // Skip intrinsic ID
2319
2320 auto GetMultiVecOperand = [&]() {
2321 SmallVector<SDValue, 4> Regs(OpsIter, OpsIter + NumVecs);
2322 OpsIter += NumVecs;
2323 return createZMulTuple(Regs);
2324 };
2325
2326 if (HasPred)
2327 Ops.push_back(*OpsIter++);
2328
2329 Ops.push_back(GetMultiVecOperand());
2330 if (IsZmMulti)
2331 Ops.push_back(GetMultiVecOperand());
2332 else
2333 Ops.push_back(*OpsIter++);
2334
2335 // Append any remaining operands.
2336 Ops.append(OpsIter, N->op_end());
2337 SDNode *Intrinsic;
2338 Intrinsic = CurDAG->getMachineNode(Opcode, DL, MVT::Untyped, Ops);
2339 SDValue SuperReg = SDValue(Intrinsic, 0);
2340 for (unsigned i = 0; i < NumVecs; ++i)
2341 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2342 AArch64::zsub0 + i, DL, VT, SuperReg));
2343
2344 CurDAG->RemoveDeadNode(N);
2345}
2346
2347void AArch64DAGToDAGISel::SelectPredicatedLoad(SDNode *N, unsigned NumVecs,
2348 unsigned Scale, unsigned Opc_ri,
2349 unsigned Opc_rr, bool IsIntr) {
2350 assert(Scale < 5 && "Invalid scaling value.");
2351 SDLoc DL(N);
2352 EVT VT = N->getValueType(0);
2353 SDValue Chain = N->getOperand(0);
2354
2355 // Optimize addressing mode.
2356 SDValue Base, Offset;
2357 unsigned Opc;
2358 std::tie(Opc, Base, Offset) = findAddrModeSVELoadStore(
2359 N, Opc_rr, Opc_ri, N->getOperand(IsIntr ? 3 : 2),
2360 CurDAG->getTargetConstant(0, DL, MVT::i64), Scale);
2361
2362 SDValue Ops[] = {N->getOperand(IsIntr ? 2 : 1), // Predicate
2363 Base, // Memory operand
2364 Offset, Chain};
2365
2366 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2367
2368 SDNode *Load = CurDAG->getMachineNode(Opc, DL, ResTys, Ops);
2369 SDValue SuperReg = SDValue(Load, 0);
2370 for (unsigned i = 0; i < NumVecs; ++i)
2371 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2372 AArch64::zsub0 + i, DL, VT, SuperReg));
2373
2374 // Copy chain
2375 unsigned ChainIdx = NumVecs;
2376 ReplaceUses(SDValue(N, ChainIdx), SDValue(Load, 1));
2377 CurDAG->RemoveDeadNode(N);
2378}
2379
2380void AArch64DAGToDAGISel::SelectContiguousMultiVectorLoad(SDNode *N,
2381 unsigned NumVecs,
2382 unsigned Scale,
2383 unsigned Opc_ri,
2384 unsigned Opc_rr) {
2385 assert(Scale < 4 && "Invalid scaling value.");
2386 SDLoc DL(N);
2387 EVT VT = N->getValueType(0);
2388 SDValue Chain = N->getOperand(0);
2389
2390 SDValue PNg = N->getOperand(2);
2391 SDValue Base = N->getOperand(3);
2392 SDValue Offset = CurDAG->getTargetConstant(0, DL, MVT::i64);
2393 unsigned Opc;
2394 std::tie(Opc, Base, Offset) =
2395 findAddrModeSVELoadStore(N, Opc_rr, Opc_ri, Base, Offset, Scale);
2396
2397 SDValue Ops[] = {PNg, // Predicate-as-counter
2398 Base, // Memory operand
2399 Offset, Chain};
2400
2401 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2402
2403 SDNode *Load = CurDAG->getMachineNode(Opc, DL, ResTys, Ops);
2404 SDValue SuperReg = SDValue(Load, 0);
2405 for (unsigned i = 0; i < NumVecs; ++i)
2406 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2407 AArch64::zsub0 + i, DL, VT, SuperReg));
2408
2409 // Copy chain
2410 unsigned ChainIdx = NumVecs;
2411 ReplaceUses(SDValue(N, ChainIdx), SDValue(Load, 1));
2412 CurDAG->RemoveDeadNode(N);
2413}
2414
2415void AArch64DAGToDAGISel::SelectFrintFromVT(SDNode *N, unsigned NumVecs,
2416 unsigned Opcode) {
2417 if (N->getValueType(0) != MVT::nxv4f32)
2418 return;
2419 SelectUnaryMultiIntrinsic(N, NumVecs, true, Opcode);
2420}
2421
2422void AArch64DAGToDAGISel::SelectMultiVectorLutiLane(SDNode *Node,
2423 unsigned NumOutVecs,
2424 unsigned Opc,
2425 uint32_t MaxImm) {
2426 if (ConstantSDNode *Imm = dyn_cast<ConstantSDNode>(Node->getOperand(4)))
2427 if (Imm->getZExtValue() > MaxImm)
2428 return;
2429
2430 SDValue ZtValue;
2431 if (!ImmToReg<AArch64::ZT0, 0>(Node->getOperand(2), ZtValue))
2432 return;
2433
2434 SDValue Chain = Node->getOperand(0);
2435 SDValue Ops[] = {ZtValue, Node->getOperand(3), Node->getOperand(4), Chain};
2436 SDLoc DL(Node);
2437 EVT VT = Node->getValueType(0);
2438
2439 SDNode *Instruction =
2440 CurDAG->getMachineNode(Opc, DL, {MVT::Untyped, MVT::Other}, Ops);
2441 SDValue SuperReg = SDValue(Instruction, 0);
2442
2443 for (unsigned I = 0; I < NumOutVecs; ++I)
2444 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2445 AArch64::zsub0 + I, DL, VT, SuperReg));
2446
2447 // Copy chain
2448 unsigned ChainIdx = NumOutVecs;
2449 ReplaceUses(SDValue(Node, ChainIdx), SDValue(Instruction, 1));
2450 CurDAG->RemoveDeadNode(Node);
2451}
2452
2453void AArch64DAGToDAGISel::SelectMultiVectorLuti6LaneX4(SDNode *Node,
2454 unsigned NumIndexVecs) {
2455 assert((NumIndexVecs == 2 || NumIndexVecs == 3) &&
2456 "unexpected number of index vectors");
2457
2458 constexpr unsigned FirstIndexOp = 3;
2459 unsigned ImmOp = FirstIndexOp + NumIndexVecs;
2460 auto *Imm = dyn_cast<ConstantSDNode>(Node->getOperand(ImmOp));
2461 if (!Imm || Imm->getZExtValue() > 1)
2462 return;
2463
2464 // The luti6 instruction always takes a 2-register Zm index tuple. The x3
2465 // ACLE form provides three index vectors, so the lane selects which adjacent
2466 // pair to use before forming Zm (op 3/4 or op 4/5, with op6 as imm)
2467 unsigned Lane = Imm->getZExtValue();
2468 unsigned IndexOp = FirstIndexOp;
2469 if (NumIndexVecs == 3)
2470 IndexOp += Lane;
2471
2472 SDValue TableTuple = createZTuple({Node->getOperand(1), Node->getOperand(2)});
2473 SDValue IndexTuple =
2474 createZTuple({Node->getOperand(IndexOp), Node->getOperand(IndexOp + 1)});
2475 SDValue Ops[] = {TableTuple, IndexTuple, Node->getOperand(ImmOp)};
2476
2477 SDLoc DL(Node);
2478 EVT VT = Node->getValueType(0);
2479 SDNode *Instruction =
2480 CurDAG->getMachineNode(AArch64::LUTI6_4Z2Z2ZI, DL, MVT::Untyped, Ops);
2481 SDValue SuperReg = SDValue(Instruction, 0);
2482
2483 for (unsigned I = 0; I < 4; ++I)
2484 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2485 AArch64::zsub0 + I, DL, VT, SuperReg));
2486
2487 CurDAG->RemoveDeadNode(Node);
2488}
2489
2490void AArch64DAGToDAGISel::SelectMultiVectorLuti(SDNode *Node,
2491 unsigned NumOutVecs,
2492 unsigned Opc,
2493 unsigned NumInVecs) {
2494 assert((NumInVecs == 2 || NumInVecs == 3) &&
2495 "unexpected number of input vectors");
2496
2497 SDValue ZtValue;
2498 if (!ImmToReg<AArch64::ZT0, 0>(Node->getOperand(2), ZtValue))
2499 return;
2500
2501 SmallVector<SDValue, 4> Regs(Node->ops().slice(3, NumInVecs));
2502 SDValue ZTuple = NumInVecs == 3 ? createZTuple(Regs) : createZMulTuple(Regs);
2503 SDValue Ops[] = {ZtValue, ZTuple, Node->getOperand(0)};
2504
2505 SDLoc DL(Node);
2506 EVT VT = Node->getValueType(0);
2507
2508 SDNode *Instruction =
2509 CurDAG->getMachineNode(Opc, DL, {MVT::Untyped, MVT::Other}, Ops);
2510 SDValue SuperReg = SDValue(Instruction, 0);
2511
2512 for (unsigned I = 0; I < NumOutVecs; ++I)
2513 ReplaceUses(SDValue(Node, I), CurDAG->getTargetExtractSubreg(
2514 AArch64::zsub0 + I, DL, VT, SuperReg));
2515
2516 ReplaceUses(SDValue(Node, NumOutVecs), SDValue(Instruction, 1));
2517 CurDAG->RemoveDeadNode(Node);
2518}
2519
2520void AArch64DAGToDAGISel::SelectClamp(SDNode *N, unsigned NumVecs,
2521 unsigned Op) {
2522 SDLoc DL(N);
2523 EVT VT = N->getValueType(0);
2524
2525 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2526 SDValue Zd = createZMulTuple(Regs);
2527 SDValue Zn = N->getOperand(1 + NumVecs);
2528 SDValue Zm = N->getOperand(2 + NumVecs);
2529
2530 SDValue Ops[] = {Zd, Zn, Zm};
2531
2532 SDNode *Intrinsic = CurDAG->getMachineNode(Op, DL, MVT::Untyped, Ops);
2533 SDValue SuperReg = SDValue(Intrinsic, 0);
2534 for (unsigned i = 0; i < NumVecs; ++i)
2535 ReplaceUses(SDValue(N, i), CurDAG->getTargetExtractSubreg(
2536 AArch64::zsub0 + i, DL, VT, SuperReg));
2537
2538 CurDAG->RemoveDeadNode(N);
2539}
2540
2541bool SelectSMETile(unsigned &BaseReg, unsigned TileNum) {
2542 switch (BaseReg) {
2543 default:
2544 return false;
2545 case AArch64::ZA:
2546 case AArch64::ZAB0:
2547 if (TileNum == 0)
2548 break;
2549 return false;
2550 case AArch64::ZAH0:
2551 if (TileNum <= 1)
2552 break;
2553 return false;
2554 case AArch64::ZAS0:
2555 if (TileNum <= 3)
2556 break;
2557 return false;
2558 case AArch64::ZAD0:
2559 if (TileNum <= 7)
2560 break;
2561 return false;
2562 }
2563
2564 BaseReg += TileNum;
2565 return true;
2566}
2567
2568template <unsigned MaxIdx, unsigned Scale>
2569void AArch64DAGToDAGISel::SelectMultiVectorMove(SDNode *N, unsigned NumVecs,
2570 unsigned BaseReg, unsigned Op) {
2571 unsigned TileNum = 0;
2572 if (BaseReg != AArch64::ZA)
2573 TileNum = N->getConstantOperandVal(2);
2574
2575 if (!SelectSMETile(BaseReg, TileNum))
2576 return;
2577
2578 SDValue SliceBase, Base, Offset;
2579 if (BaseReg == AArch64::ZA)
2580 SliceBase = N->getOperand(2);
2581 else
2582 SliceBase = N->getOperand(3);
2583
2584 if (!SelectSMETileSlice(SliceBase, MaxIdx, Base, Offset, Scale))
2585 return;
2586
2587 SDLoc DL(N);
2588 SDValue SubReg = CurDAG->getRegister(BaseReg, MVT::Other);
2589 SDValue Ops[] = {SubReg, Base, Offset, /*Chain*/ N->getOperand(0)};
2590 SDNode *Mov = CurDAG->getMachineNode(Op, DL, {MVT::Untyped, MVT::Other}, Ops);
2591
2592 EVT VT = N->getValueType(0);
2593 for (unsigned I = 0; I < NumVecs; ++I)
2594 ReplaceUses(SDValue(N, I),
2595 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
2596 SDValue(Mov, 0)));
2597 // Copy chain
2598 unsigned ChainIdx = NumVecs;
2599 ReplaceUses(SDValue(N, ChainIdx), SDValue(Mov, 1));
2600 CurDAG->RemoveDeadNode(N);
2601}
2602
2603void AArch64DAGToDAGISel::SelectMultiVectorMoveZ(SDNode *N, unsigned NumVecs,
2604 unsigned Op, unsigned MaxIdx,
2605 unsigned Scale, unsigned BaseReg) {
2606 // Slice can be in different positions
2607 // The array to vector: llvm.aarch64.sme.readz.<h/v>.<sz>(slice)
2608 // The tile to vector: llvm.aarch64.sme.readz.<h/v>.<sz>(tile, slice)
2609 SDValue SliceBase = N->getOperand(2);
2610 if (BaseReg != AArch64::ZA)
2611 SliceBase = N->getOperand(3);
2612
2613 SDValue Base, Offset;
2614 if (!SelectSMETileSlice(SliceBase, MaxIdx, Base, Offset, Scale))
2615 return;
2616 // The correct Za tile number is computed in Machine Instruction
2617 // See EmitZAInstr
2618 // DAG cannot select Za tile as an output register with ZReg
2619 SDLoc DL(N);
2621 if (BaseReg != AArch64::ZA )
2622 Ops.push_back(N->getOperand(2));
2623 Ops.push_back(Base);
2624 Ops.push_back(Offset);
2625 Ops.push_back(N->getOperand(0)); //Chain
2626 SDNode *Mov = CurDAG->getMachineNode(Op, DL, {MVT::Untyped, MVT::Other}, Ops);
2627
2628 EVT VT = N->getValueType(0);
2629 for (unsigned I = 0; I < NumVecs; ++I)
2630 ReplaceUses(SDValue(N, I),
2631 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
2632 SDValue(Mov, 0)));
2633
2634 // Copy chain
2635 unsigned ChainIdx = NumVecs;
2636 ReplaceUses(SDValue(N, ChainIdx), SDValue(Mov, 1));
2637 CurDAG->RemoveDeadNode(N);
2638}
2639
2640void AArch64DAGToDAGISel::SelectUnaryMultiIntrinsic(SDNode *N,
2641 unsigned NumOutVecs,
2642 bool IsTupleInput,
2643 unsigned Opc) {
2644 SDLoc DL(N);
2645 EVT VT = N->getValueType(0);
2646 unsigned NumInVecs = N->getNumOperands() - 1;
2647
2649 if (IsTupleInput) {
2650 assert((NumInVecs == 2 || NumInVecs == 4) &&
2651 "Don't know how to handle multi-register input!");
2652 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumInVecs));
2653 Ops.push_back(createZMulTuple(Regs));
2654 } else {
2655 // All intrinsic nodes have the ID as the first operand, hence the "1 + I".
2656 for (unsigned I = 0; I < NumInVecs; I++)
2657 Ops.push_back(N->getOperand(1 + I));
2658 }
2659
2660 SDNode *Res = CurDAG->getMachineNode(Opc, DL, MVT::Untyped, Ops);
2661 SDValue SuperReg = SDValue(Res, 0);
2662
2663 for (unsigned I = 0; I < NumOutVecs; I++)
2664 ReplaceUses(SDValue(N, I), CurDAG->getTargetExtractSubreg(
2665 AArch64::zsub0 + I, DL, VT, SuperReg));
2666 CurDAG->RemoveDeadNode(N);
2667}
2668
2669void AArch64DAGToDAGISel::SelectStore(SDNode *N, unsigned NumVecs,
2670 unsigned Opc) {
2671 SDLoc dl(N);
2672 EVT VT = N->getOperand(2)->getValueType(0);
2673
2674 // Form a REG_SEQUENCE to force register allocation.
2675 bool Is128Bit = VT.getSizeInBits() == 128;
2676 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2677 SDValue RegSeq = Is128Bit ? createQTuple(Regs) : createDTuple(Regs);
2678
2679 SDValue Ops[] = {RegSeq, N->getOperand(NumVecs + 2), N->getOperand(0)};
2680 SDNode *St = CurDAG->getMachineNode(Opc, dl, N->getValueType(0), Ops);
2681
2682 // Transfer memoperands.
2683 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2684 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2685
2686 ReplaceNode(N, St);
2687}
2688
2689void AArch64DAGToDAGISel::SelectPredicatedStore(SDNode *N, unsigned NumVecs,
2690 unsigned Scale, unsigned Opc_rr,
2691 unsigned Opc_ri) {
2692 SDLoc dl(N);
2693
2694 // Form a REG_SEQUENCE to force register allocation.
2695 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2696 SDValue RegSeq = createZTuple(Regs);
2697
2698 // Optimize addressing mode.
2699 unsigned Opc;
2700 SDValue Offset, Base;
2701 std::tie(Opc, Base, Offset) = findAddrModeSVELoadStore(
2702 N, Opc_rr, Opc_ri, N->getOperand(NumVecs + 3),
2703 CurDAG->getTargetConstant(0, dl, MVT::i64), Scale);
2704
2705 SDValue Ops[] = {RegSeq, N->getOperand(NumVecs + 2), // predicate
2706 Base, // address
2707 Offset, // offset
2708 N->getOperand(0)}; // chain
2709 SDNode *St = CurDAG->getMachineNode(Opc, dl, N->getValueType(0), Ops);
2710
2711 // Transfer memoperands.
2712 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2713 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2714
2715 ReplaceNode(N, St);
2716}
2717
2718void AArch64DAGToDAGISel::SelectPostStore(SDNode *N, unsigned NumVecs,
2719 unsigned Opc) {
2720 SDLoc dl(N);
2721 EVT VT = N->getOperand(2)->getValueType(0);
2722 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2723 MVT::Other}; // Type for the Chain
2724
2725 // Form a REG_SEQUENCE to force register allocation.
2726 bool Is128Bit = VT.getSizeInBits() == 128;
2727 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2728 SDValue RegSeq = Is128Bit ? createQTuple(Regs) : createDTuple(Regs);
2729
2730 SDValue Ops[] = {RegSeq,
2731 N->getOperand(NumVecs + 1), // base register
2732 N->getOperand(NumVecs + 2), // Incremental
2733 N->getOperand(0)}; // Chain
2734 SDNode *St = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2735
2736 // Transfer memoperands.
2737 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2738 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2739
2740 ReplaceNode(N, St);
2741}
2742
2743namespace {
2744/// WidenVector - Given a value in the V64 register class, produce the
2745/// equivalent value in the V128 register class.
2746class WidenVector {
2747 SelectionDAG &DAG;
2748
2749public:
2750 WidenVector(SelectionDAG &DAG) : DAG(DAG) {}
2751
2752 SDValue operator()(SDValue V64Reg) {
2753 EVT VT = V64Reg.getValueType();
2754 unsigned NarrowSize = VT.getVectorNumElements();
2755 MVT EltTy = VT.getVectorElementType().getSimpleVT();
2756 MVT WideTy = MVT::getVectorVT(EltTy, 2 * NarrowSize);
2757 SDLoc DL(V64Reg);
2758
2759 SDValue Undef =
2760 SDValue(DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, WideTy), 0);
2761 return DAG.getTargetInsertSubreg(AArch64::dsub, DL, WideTy, Undef, V64Reg);
2762 }
2763};
2764} // namespace
2765
2766/// NarrowVector - Given a value in the V128 register class, produce the
2767/// equivalent value in the V64 register class.
2769 EVT VT = V128Reg.getValueType();
2770 unsigned WideSize = VT.getVectorNumElements();
2771 MVT EltTy = VT.getVectorElementType().getSimpleVT();
2772 MVT NarrowTy = MVT::getVectorVT(EltTy, WideSize / 2);
2773
2774 return DAG.getTargetExtractSubreg(AArch64::dsub, SDLoc(V128Reg), NarrowTy,
2775 V128Reg);
2776}
2777
2778void AArch64DAGToDAGISel::SelectLoadLane(SDNode *N, unsigned NumVecs,
2779 unsigned Opc) {
2780 SDLoc dl(N);
2781 EVT VT = N->getValueType(0);
2782 bool Narrow = VT.getSizeInBits() == 64;
2783
2784 // Form a REG_SEQUENCE to force register allocation.
2785 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2786
2787 if (Narrow)
2788 transform(Regs, Regs.begin(),
2789 WidenVector(*CurDAG));
2790
2791 SDValue RegSeq = createQTuple(Regs);
2792
2793 const EVT ResTys[] = {MVT::Untyped, MVT::Other};
2794
2795 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 2);
2796
2797 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2798 N->getOperand(NumVecs + 3), N->getOperand(0)};
2799 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2800 SDValue SuperReg = SDValue(Ld, 0);
2801
2802 EVT WideVT = RegSeq.getOperand(1)->getValueType(0);
2803 static const unsigned QSubs[] = { AArch64::qsub0, AArch64::qsub1,
2804 AArch64::qsub2, AArch64::qsub3 };
2805 for (unsigned i = 0; i < NumVecs; ++i) {
2806 SDValue NV = CurDAG->getTargetExtractSubreg(QSubs[i], dl, WideVT, SuperReg);
2807 if (Narrow)
2808 NV = NarrowVector(NV, *CurDAG);
2809 ReplaceUses(SDValue(N, i), NV);
2810 }
2811
2812 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 1));
2813 CurDAG->RemoveDeadNode(N);
2814}
2815
2816void AArch64DAGToDAGISel::SelectPostLoadLane(SDNode *N, unsigned NumVecs,
2817 unsigned Opc) {
2818 SDLoc dl(N);
2819 EVT VT = N->getValueType(0);
2820 bool Narrow = VT.getSizeInBits() == 64;
2821
2822 // Form a REG_SEQUENCE to force register allocation.
2823 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2824
2825 if (Narrow)
2826 transform(Regs, Regs.begin(),
2827 WidenVector(*CurDAG));
2828
2829 SDValue RegSeq = createQTuple(Regs);
2830
2831 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2832 RegSeq->getValueType(0), MVT::Other};
2833
2834 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 1);
2835
2836 SDValue Ops[] = {RegSeq,
2837 CurDAG->getTargetConstant(LaneNo, dl,
2838 MVT::i64), // Lane Number
2839 N->getOperand(NumVecs + 2), // Base register
2840 N->getOperand(NumVecs + 3), // Incremental
2841 N->getOperand(0)};
2842 SDNode *Ld = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2843
2844 // Update uses of the write back register
2845 ReplaceUses(SDValue(N, NumVecs), SDValue(Ld, 0));
2846
2847 // Update uses of the vector list
2848 SDValue SuperReg = SDValue(Ld, 1);
2849 if (NumVecs == 1) {
2850 ReplaceUses(SDValue(N, 0),
2851 Narrow ? NarrowVector(SuperReg, *CurDAG) : SuperReg);
2852 } else {
2853 EVT WideVT = RegSeq.getOperand(1)->getValueType(0);
2854 static const unsigned QSubs[] = { AArch64::qsub0, AArch64::qsub1,
2855 AArch64::qsub2, AArch64::qsub3 };
2856 for (unsigned i = 0; i < NumVecs; ++i) {
2857 SDValue NV = CurDAG->getTargetExtractSubreg(QSubs[i], dl, WideVT,
2858 SuperReg);
2859 if (Narrow)
2860 NV = NarrowVector(NV, *CurDAG);
2861 ReplaceUses(SDValue(N, i), NV);
2862 }
2863 }
2864
2865 // Update the Chain
2866 ReplaceUses(SDValue(N, NumVecs + 1), SDValue(Ld, 2));
2867 CurDAG->RemoveDeadNode(N);
2868}
2869
2870void AArch64DAGToDAGISel::SelectStoreLane(SDNode *N, unsigned NumVecs,
2871 unsigned Opc) {
2872 SDLoc dl(N);
2873 EVT VT = N->getOperand(2)->getValueType(0);
2874 bool Narrow = VT.getSizeInBits() == 64;
2875
2876 // Form a REG_SEQUENCE to force register allocation.
2877 SmallVector<SDValue, 4> Regs(N->ops().slice(2, NumVecs));
2878
2879 if (Narrow)
2880 transform(Regs, Regs.begin(),
2881 WidenVector(*CurDAG));
2882
2883 SDValue RegSeq = createQTuple(Regs);
2884
2885 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 2);
2886
2887 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2888 N->getOperand(NumVecs + 3), N->getOperand(0)};
2889 SDNode *St = CurDAG->getMachineNode(Opc, dl, MVT::Other, Ops);
2890
2891 // Transfer memoperands.
2892 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2893 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2894
2895 ReplaceNode(N, St);
2896}
2897
2898void AArch64DAGToDAGISel::SelectPostStoreLane(SDNode *N, unsigned NumVecs,
2899 unsigned Opc) {
2900 SDLoc dl(N);
2901 EVT VT = N->getOperand(2)->getValueType(0);
2902 bool Narrow = VT.getSizeInBits() == 64;
2903
2904 // Form a REG_SEQUENCE to force register allocation.
2905 SmallVector<SDValue, 4> Regs(N->ops().slice(1, NumVecs));
2906
2907 if (Narrow)
2908 transform(Regs, Regs.begin(),
2909 WidenVector(*CurDAG));
2910
2911 SDValue RegSeq = createQTuple(Regs);
2912
2913 const EVT ResTys[] = {MVT::i64, // Type of the write back register
2914 MVT::Other};
2915
2916 unsigned LaneNo = N->getConstantOperandVal(NumVecs + 1);
2917
2918 SDValue Ops[] = {RegSeq, CurDAG->getTargetConstant(LaneNo, dl, MVT::i64),
2919 N->getOperand(NumVecs + 2), // Base Register
2920 N->getOperand(NumVecs + 3), // Incremental
2921 N->getOperand(0)};
2922 SDNode *St = CurDAG->getMachineNode(Opc, dl, ResTys, Ops);
2923
2924 // Transfer memoperands.
2925 MachineMemOperand *MemOp = cast<MemIntrinsicSDNode>(N)->getMemOperand();
2926 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
2927
2928 ReplaceNode(N, St);
2929}
2930
2932 unsigned &Opc, SDValue &Opd0,
2933 unsigned &LSB, unsigned &MSB,
2934 unsigned NumberOfIgnoredLowBits,
2935 bool BiggerPattern) {
2936 assert(N->getOpcode() == ISD::AND &&
2937 "N must be a AND operation to call this function");
2938
2939 EVT VT = N->getValueType(0);
2940
2941 // Here we can test the type of VT and return false when the type does not
2942 // match, but since it is done prior to that call in the current context
2943 // we turned that into an assert to avoid redundant code.
2944 assert((VT == MVT::i32 || VT == MVT::i64) &&
2945 "Type checking must have been done before calling this function");
2946
2947 // FIXME: simplify-demanded-bits in DAGCombine will probably have
2948 // changed the AND node to a 32-bit mask operation. We'll have to
2949 // undo that as part of the transform here if we want to catch all
2950 // the opportunities.
2951 // Currently the NumberOfIgnoredLowBits argument helps to recover
2952 // from these situations when matching bigger pattern (bitfield insert).
2953
2954 // For unsigned extracts, check for a shift right and mask
2955 uint64_t AndImm = 0;
2956 if (!isOpcWithIntImmediate(N, ISD::AND, AndImm))
2957 return false;
2958
2959 const SDNode *Op0 = N->getOperand(0).getNode();
2960
2961 // Because of simplify-demanded-bits in DAGCombine, the mask may have been
2962 // simplified. Try to undo that
2963 AndImm |= maskTrailingOnes<uint64_t>(NumberOfIgnoredLowBits);
2964
2965 // The immediate is a mask of the low bits iff imm & (imm+1) == 0
2966 if (AndImm & (AndImm + 1))
2967 return false;
2968
2969 bool ClampMSB = false;
2970 uint64_t SrlImm = 0;
2971 // Handle the SRL + ANY_EXTEND case.
2972 if (VT == MVT::i64 && Op0->getOpcode() == ISD::ANY_EXTEND &&
2973 isOpcWithIntImmediate(Op0->getOperand(0).getNode(), ISD::SRL, SrlImm)) {
2974 // Extend the incoming operand of the SRL to 64-bit.
2975 Opd0 = Widen(CurDAG, Op0->getOperand(0).getOperand(0));
2976 // Make sure to clamp the MSB so that we preserve the semantics of the
2977 // original operations.
2978 ClampMSB = true;
2979 } else if (VT == MVT::i32 && Op0->getOpcode() == ISD::TRUNCATE &&
2981 SrlImm)) {
2982 // If the shift result was truncated, we can still combine them.
2983 Opd0 = Op0->getOperand(0).getOperand(0);
2984
2985 // Use the type of SRL node.
2986 VT = Opd0->getValueType(0);
2987 } else if (isOpcWithIntImmediate(Op0, ISD::SRL, SrlImm)) {
2988 Opd0 = Op0->getOperand(0);
2989 ClampMSB = (VT == MVT::i32);
2990 } else if (BiggerPattern) {
2991 // Let's pretend a 0 shift right has been performed.
2992 // The resulting code will be at least as good as the original one
2993 // plus it may expose more opportunities for bitfield insert pattern.
2994 // FIXME: Currently we limit this to the bigger pattern, because
2995 // some optimizations expect AND and not UBFM.
2996 Opd0 = N->getOperand(0);
2997 } else
2998 return false;
2999
3000 // Bail out on large immediates. This happens when no proper
3001 // combining/constant folding was performed.
3002 if (!BiggerPattern && (SrlImm <= 0 || SrlImm >= VT.getSizeInBits())) {
3003 LLVM_DEBUG(
3004 (dbgs() << N
3005 << ": Found large shift immediate, this should not happen\n"));
3006 return false;
3007 }
3008
3009 LSB = SrlImm;
3010 MSB = SrlImm +
3011 (VT == MVT::i32 ? llvm::countr_one<uint32_t>(AndImm)
3012 : llvm::countr_one<uint64_t>(AndImm)) -
3013 1;
3014 if (ClampMSB)
3015 // Since we're moving the extend before the right shift operation, we need
3016 // to clamp the MSB to make sure we don't shift in undefined bits instead of
3017 // the zeros which would get shifted in with the original right shift
3018 // operation.
3019 MSB = MSB > 31 ? 31 : MSB;
3020
3021 Opc = VT == MVT::i32 ? AArch64::UBFMWri : AArch64::UBFMXri;
3022 return true;
3023}
3024
3026 SDValue &Opd0, unsigned &Immr,
3027 unsigned &Imms) {
3028 assert(N->getOpcode() == ISD::SIGN_EXTEND_INREG);
3029
3030 EVT VT = N->getValueType(0);
3031 unsigned BitWidth = VT.getSizeInBits();
3032 assert((VT == MVT::i32 || VT == MVT::i64) &&
3033 "Type checking must have been done before calling this function");
3034
3035 SDValue Op = N->getOperand(0);
3036 if (Op->getOpcode() == ISD::TRUNCATE) {
3037 Op = Op->getOperand(0);
3038 VT = Op->getValueType(0);
3039 BitWidth = VT.getSizeInBits();
3040 }
3041
3042 uint64_t ShiftImm;
3043 if (!isOpcWithIntImmediate(Op.getNode(), ISD::SRL, ShiftImm) &&
3044 !isOpcWithIntImmediate(Op.getNode(), ISD::SRA, ShiftImm))
3045 return false;
3046
3047 unsigned Width = cast<VTSDNode>(N->getOperand(1))->getVT().getSizeInBits();
3048 if (ShiftImm + Width > BitWidth)
3049 return false;
3050
3051 Opc = (VT == MVT::i32) ? AArch64::SBFMWri : AArch64::SBFMXri;
3052 Opd0 = Op.getOperand(0);
3053 Immr = ShiftImm;
3054 Imms = ShiftImm + Width - 1;
3055 return true;
3056}
3057
3059 SDValue &Opd0, unsigned &LSB,
3060 unsigned &MSB) {
3061 // We are looking for the following pattern which basically extracts several
3062 // continuous bits from the source value and places it from the LSB of the
3063 // destination value, all other bits of the destination value or set to zero:
3064 //
3065 // Value2 = AND Value, MaskImm
3066 // SRL Value2, ShiftImm
3067 //
3068 // with MaskImm >> ShiftImm to search for the bit width.
3069 //
3070 // This gets selected into a single UBFM:
3071 //
3072 // UBFM Value, ShiftImm, Log2_64(MaskImm)
3073 //
3074
3075 if (N->getOpcode() != ISD::SRL)
3076 return false;
3077
3078 uint64_t AndMask = 0;
3079 if (!isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::AND, AndMask))
3080 return false;
3081
3082 Opd0 = N->getOperand(0).getOperand(0);
3083
3084 uint64_t SrlImm = 0;
3085 if (!isIntImmediate(N->getOperand(1), SrlImm))
3086 return false;
3087
3088 // Check whether we really have several bits extract here.
3089 if (!isMask_64(AndMask >> SrlImm))
3090 return false;
3091
3092 Opc = N->getValueType(0) == MVT::i32 ? AArch64::UBFMWri : AArch64::UBFMXri;
3093 LSB = SrlImm;
3094 MSB = llvm::Log2_64(AndMask);
3095 return true;
3096}
3097
3098static bool isBitfieldExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0,
3099 unsigned &Immr, unsigned &Imms,
3100 bool BiggerPattern) {
3101 assert((N->getOpcode() == ISD::SRA || N->getOpcode() == ISD::SRL) &&
3102 "N must be a SHR/SRA operation to call this function");
3103
3104 EVT VT = N->getValueType(0);
3105
3106 // Here we can test the type of VT and return false when the type does not
3107 // match, but since it is done prior to that call in the current context
3108 // we turned that into an assert to avoid redundant code.
3109 assert((VT == MVT::i32 || VT == MVT::i64) &&
3110 "Type checking must have been done before calling this function");
3111
3112 // Check for AND + SRL doing several bits extract.
3113 if (isSeveralBitsExtractOpFromShr(N, Opc, Opd0, Immr, Imms))
3114 return true;
3115
3116 // We're looking for a shift of a shift.
3117 uint64_t ShlImm = 0;
3118 uint64_t TruncBits = 0;
3119 if (isOpcWithIntImmediate(N->getOperand(0).getNode(), ISD::SHL, ShlImm)) {
3120 Opd0 = N->getOperand(0).getOperand(0);
3121 } else if (VT == MVT::i32 && N->getOpcode() == ISD::SRL &&
3122 N->getOperand(0).getNode()->getOpcode() == ISD::TRUNCATE) {
3123 // We are looking for a shift of truncate. Truncate from i64 to i32 could
3124 // be considered as setting high 32 bits as zero. Our strategy here is to
3125 // always generate 64bit UBFM. This consistency will help the CSE pass
3126 // later find more redundancy.
3127 Opd0 = N->getOperand(0).getOperand(0);
3128 TruncBits = Opd0->getValueType(0).getSizeInBits() - VT.getSizeInBits();
3129 VT = Opd0.getValueType();
3130 assert(VT == MVT::i64 && "the promoted type should be i64");
3131 } else if (BiggerPattern) {
3132 // Let's pretend a 0 shift left has been performed.
3133 // FIXME: Currently we limit this to the bigger pattern case,
3134 // because some optimizations expect AND and not UBFM
3135 Opd0 = N->getOperand(0);
3136 } else
3137 return false;
3138
3139 // Missing combines/constant folding may have left us with strange
3140 // constants.
3141 if (ShlImm >= VT.getSizeInBits()) {
3142 LLVM_DEBUG(
3143 (dbgs() << N
3144 << ": Found large shift immediate, this should not happen\n"));
3145 return false;
3146 }
3147
3148 uint64_t SrlImm = 0;
3149 if (!isIntImmediate(N->getOperand(1), SrlImm))
3150 return false;
3151
3152 assert(SrlImm > 0 && SrlImm < VT.getSizeInBits() &&
3153 "bad amount in shift node!");
3154 int immr = SrlImm - ShlImm;
3155 Immr = immr < 0 ? immr + VT.getSizeInBits() : immr;
3156 Imms = VT.getSizeInBits() - ShlImm - TruncBits - 1;
3157 // SRA requires a signed extraction
3158 if (VT == MVT::i32)
3159 Opc = N->getOpcode() == ISD::SRA ? AArch64::SBFMWri : AArch64::UBFMWri;
3160 else
3161 Opc = N->getOpcode() == ISD::SRA ? AArch64::SBFMXri : AArch64::UBFMXri;
3162 return true;
3163}
3164
3165static bool isBitfieldExtractOp(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc,
3166 SDValue &Opd0, unsigned &Immr, unsigned &Imms,
3167 unsigned NumberOfIgnoredLowBits = 0,
3168 bool BiggerPattern = false) {
3169 if (N->getValueType(0) != MVT::i32 && N->getValueType(0) != MVT::i64)
3170 return false;
3171
3172 switch (N->getOpcode()) {
3173 default:
3174 if (!N->isMachineOpcode())
3175 return false;
3176 break;
3177 case ISD::AND:
3178 return isBitfieldExtractOpFromAnd(CurDAG, N, Opc, Opd0, Immr, Imms,
3179 NumberOfIgnoredLowBits, BiggerPattern);
3180 case ISD::SRL:
3181 case ISD::SRA:
3182 return isBitfieldExtractOpFromShr(N, Opc, Opd0, Immr, Imms, BiggerPattern);
3183
3185 return isBitfieldExtractOpFromSExtInReg(N, Opc, Opd0, Immr, Imms);
3186 }
3187
3188 unsigned NOpc = N->getMachineOpcode();
3189 switch (NOpc) {
3190 default:
3191 return false;
3192 case AArch64::SBFMWri:
3193 case AArch64::UBFMWri:
3194 case AArch64::SBFMXri:
3195 case AArch64::UBFMXri:
3196 Opc = NOpc;
3197 Opd0 = N->getOperand(0);
3198 Immr = N->getConstantOperandVal(1);
3199 Imms = N->getConstantOperandVal(2);
3200 return true;
3201 }
3202 // Unreachable
3203 return false;
3204}
3205
3206bool AArch64DAGToDAGISel::tryBitfieldExtractOp(SDNode *N) {
3207 unsigned Opc, Immr, Imms;
3208 SDValue Opd0;
3209 if (!isBitfieldExtractOp(CurDAG, N, Opc, Opd0, Immr, Imms))
3210 return false;
3211
3212 EVT VT = N->getValueType(0);
3213 SDLoc dl(N);
3214
3215 // If the bit extract operation is 64bit but the original type is 32bit, we
3216 // need to add one EXTRACT_SUBREG.
3217 if ((Opc == AArch64::SBFMXri || Opc == AArch64::UBFMXri) && VT == MVT::i32) {
3218 SDValue Ops64[] = {Opd0, CurDAG->getTargetConstant(Immr, dl, MVT::i64),
3219 CurDAG->getTargetConstant(Imms, dl, MVT::i64)};
3220
3221 SDNode *BFM = CurDAG->getMachineNode(Opc, dl, MVT::i64, Ops64);
3222 SDValue Inner = CurDAG->getTargetExtractSubreg(AArch64::sub_32, dl,
3223 MVT::i32, SDValue(BFM, 0));
3224 ReplaceNode(N, Inner.getNode());
3225 return true;
3226 }
3227
3228 SDValue Ops[] = {Opd0, CurDAG->getTargetConstant(Immr, dl, VT),
3229 CurDAG->getTargetConstant(Imms, dl, VT)};
3230 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
3231 return true;
3232}
3233
3234/// Does DstMask form a complementary pair with the mask provided by
3235/// BitsToBeInserted, suitable for use in a BFI instruction. Roughly speaking,
3236/// this asks whether DstMask zeroes precisely those bits that will be set by
3237/// the other half.
3238static bool isBitfieldDstMask(uint64_t DstMask, const APInt &BitsToBeInserted,
3239 unsigned NumberOfIgnoredHighBits, EVT VT) {
3240 assert((VT == MVT::i32 || VT == MVT::i64) &&
3241 "i32 or i64 mask type expected!");
3242 unsigned BitWidth = VT.getSizeInBits() - NumberOfIgnoredHighBits;
3243
3244 // Enable implicitTrunc as we're intentionally ignoring high bits.
3245 APInt SignificantDstMask =
3246 APInt(BitWidth, DstMask, /*isSigned=*/false, /*implicitTrunc=*/true);
3247 APInt SignificantBitsToBeInserted = BitsToBeInserted.zextOrTrunc(BitWidth);
3248
3249 return (SignificantDstMask & SignificantBitsToBeInserted) == 0 &&
3250 (SignificantDstMask | SignificantBitsToBeInserted).isAllOnes();
3251}
3252
3253// Look for bits that will be useful for later uses.
3254// A bit is consider useless as soon as it is dropped and never used
3255// before it as been dropped.
3256// E.g., looking for useful bit of x
3257// 1. y = x & 0x7
3258// 2. z = y >> 2
3259// After #1, x useful bits are 0x7, then the useful bits of x, live through
3260// y.
3261// After #2, the useful bits of x are 0x4.
3262// However, if x is used on an unpredictable instruction, then all its bits
3263// are useful.
3264// E.g.
3265// 1. y = x & 0x7
3266// 2. z = y >> 2
3267// 3. str x, [@x]
3268static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth = 0);
3269
3271 unsigned Depth) {
3272 uint64_t Imm =
3273 cast<const ConstantSDNode>(Op.getOperand(1).getNode())->getZExtValue();
3275 UsefulBits &= APInt(UsefulBits.getBitWidth(), Imm);
3276 getUsefulBits(Op, UsefulBits, Depth + 1);
3277}
3278
3280 uint64_t Imm, uint64_t MSB,
3281 unsigned Depth) {
3282 // inherit the bitwidth value
3283 APInt OpUsefulBits(UsefulBits);
3284 OpUsefulBits = 1;
3285
3286 if (MSB >= Imm) {
3287 OpUsefulBits <<= MSB - Imm + 1;
3288 --OpUsefulBits;
3289 // The interesting part will be in the lower part of the result
3290 getUsefulBits(Op, OpUsefulBits, Depth + 1);
3291 // The interesting part was starting at Imm in the argument
3292 OpUsefulBits <<= Imm;
3293 } else {
3294 OpUsefulBits <<= MSB + 1;
3295 --OpUsefulBits;
3296 // The interesting part will be shifted in the result
3297 OpUsefulBits <<= OpUsefulBits.getBitWidth() - Imm;
3298 getUsefulBits(Op, OpUsefulBits, Depth + 1);
3299 // The interesting part was at zero in the argument
3300 OpUsefulBits.lshrInPlace(OpUsefulBits.getBitWidth() - Imm);
3301 }
3302
3303 UsefulBits &= OpUsefulBits;
3304}
3305
3306static void getUsefulBitsFromUBFM(SDValue Op, APInt &UsefulBits,
3307 unsigned Depth) {
3308 uint64_t Imm =
3309 cast<const ConstantSDNode>(Op.getOperand(1).getNode())->getZExtValue();
3310 uint64_t MSB =
3311 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3312
3313 getUsefulBitsFromBitfieldMoveOpd(Op, UsefulBits, Imm, MSB, Depth);
3314}
3315
3317 unsigned Depth) {
3318 uint64_t ShiftTypeAndValue =
3319 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3320 APInt Mask(UsefulBits);
3321 Mask.clearAllBits();
3322 Mask.flipAllBits();
3323
3324 if (AArch64_AM::getShiftType(ShiftTypeAndValue) == AArch64_AM::LSL) {
3325 // Shift Left
3326 uint64_t ShiftAmt = AArch64_AM::getShiftValue(ShiftTypeAndValue);
3327 Mask <<= ShiftAmt;
3328 getUsefulBits(Op, Mask, Depth + 1);
3329 Mask.lshrInPlace(ShiftAmt);
3330 } else if (AArch64_AM::getShiftType(ShiftTypeAndValue) == AArch64_AM::LSR) {
3331 // Shift Right
3332 // We do not handle AArch64_AM::ASR, because the sign will change the
3333 // number of useful bits
3334 uint64_t ShiftAmt = AArch64_AM::getShiftValue(ShiftTypeAndValue);
3335 Mask.lshrInPlace(ShiftAmt);
3336 getUsefulBits(Op, Mask, Depth + 1);
3337 Mask <<= ShiftAmt;
3338 } else
3339 return;
3340
3341 UsefulBits &= Mask;
3342}
3343
3344static void getUsefulBitsFromBFM(SDValue Op, SDValue Orig, APInt &UsefulBits,
3345 unsigned Depth) {
3346 uint64_t Imm =
3347 cast<const ConstantSDNode>(Op.getOperand(2).getNode())->getZExtValue();
3348 uint64_t MSB =
3349 cast<const ConstantSDNode>(Op.getOperand(3).getNode())->getZExtValue();
3350
3351 APInt OpUsefulBits(UsefulBits);
3352 OpUsefulBits = 1;
3353
3354 APInt ResultUsefulBits(UsefulBits.getBitWidth(), 0);
3355 ResultUsefulBits.flipAllBits();
3356 APInt Mask(UsefulBits.getBitWidth(), 0);
3357
3358 getUsefulBits(Op, ResultUsefulBits, Depth + 1);
3359
3360 if (MSB >= Imm) {
3361 // The instruction is a BFXIL.
3362 uint64_t Width = MSB - Imm + 1;
3363 uint64_t LSB = Imm;
3364
3365 OpUsefulBits <<= Width;
3366 --OpUsefulBits;
3367
3368 if (Op.getOperand(1) == Orig) {
3369 // Copy the low bits from the result to bits starting from LSB.
3370 Mask = ResultUsefulBits & OpUsefulBits;
3371 Mask <<= LSB;
3372 }
3373
3374 if (Op.getOperand(0) == Orig)
3375 // Bits starting from LSB in the input contribute to the result.
3376 Mask |= (ResultUsefulBits & ~OpUsefulBits);
3377 } else {
3378 // The instruction is a BFI.
3379 uint64_t Width = MSB + 1;
3380 uint64_t LSB = UsefulBits.getBitWidth() - Imm;
3381
3382 OpUsefulBits <<= Width;
3383 --OpUsefulBits;
3384 OpUsefulBits <<= LSB;
3385
3386 if (Op.getOperand(1) == Orig) {
3387 // Copy the bits from the result to the zero bits.
3388 Mask = ResultUsefulBits & OpUsefulBits;
3389 Mask.lshrInPlace(LSB);
3390 }
3391
3392 if (Op.getOperand(0) == Orig)
3393 Mask |= (ResultUsefulBits & ~OpUsefulBits);
3394 }
3395
3396 UsefulBits &= Mask;
3397}
3398
3399static void getUsefulBitsForUse(SDNode *UserNode, APInt &UsefulBits,
3400 SDValue Orig, unsigned Depth) {
3401
3402 // Users of this node should have already been instruction selected
3403 // FIXME: Can we turn that into an assert?
3404 if (!UserNode->isMachineOpcode())
3405 return;
3406
3407 switch (UserNode->getMachineOpcode()) {
3408 default:
3409 return;
3410 case AArch64::ANDSWri:
3411 case AArch64::ANDSXri:
3412 case AArch64::ANDWri:
3413 case AArch64::ANDXri:
3414 // We increment Depth only when we call the getUsefulBits
3415 return getUsefulBitsFromAndWithImmediate(SDValue(UserNode, 0), UsefulBits,
3416 Depth);
3417 case AArch64::UBFMWri:
3418 case AArch64::UBFMXri:
3419 return getUsefulBitsFromUBFM(SDValue(UserNode, 0), UsefulBits, Depth);
3420
3421 case AArch64::ORRWrs:
3422 case AArch64::ORRXrs:
3423 if (UserNode->getOperand(0) != Orig && UserNode->getOperand(1) == Orig)
3424 getUsefulBitsFromOrWithShiftedReg(SDValue(UserNode, 0), UsefulBits,
3425 Depth);
3426 return;
3427 case AArch64::BFMWri:
3428 case AArch64::BFMXri:
3429 return getUsefulBitsFromBFM(SDValue(UserNode, 0), Orig, UsefulBits, Depth);
3430
3431 case AArch64::STRBBui:
3432 case AArch64::STURBBi:
3433 if (UserNode->getOperand(0) != Orig)
3434 return;
3435 UsefulBits &= APInt(UsefulBits.getBitWidth(), 0xff);
3436 return;
3437
3438 case AArch64::STRHHui:
3439 case AArch64::STURHHi:
3440 if (UserNode->getOperand(0) != Orig)
3441 return;
3442 UsefulBits &= APInt(UsefulBits.getBitWidth(), 0xffff);
3443 return;
3444 }
3445}
3446
3447static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth) {
3449 return;
3450 // Initialize UsefulBits
3451 if (!Depth) {
3452 unsigned Bitwidth = Op.getScalarValueSizeInBits();
3453 // At the beginning, assume every produced bits is useful
3454 UsefulBits = APInt(Bitwidth, 0);
3455 UsefulBits.flipAllBits();
3456 }
3457 APInt UsersUsefulBits(UsefulBits.getBitWidth(), 0);
3458
3459 for (SDNode *Node : Op.getNode()->users()) {
3460 // A use cannot produce useful bits
3461 APInt UsefulBitsForUse = APInt(UsefulBits);
3462 getUsefulBitsForUse(Node, UsefulBitsForUse, Op, Depth);
3463 UsersUsefulBits |= UsefulBitsForUse;
3464 }
3465 // UsefulBits contains the produced bits that are meaningful for the
3466 // current definition, thus a user cannot make a bit meaningful at
3467 // this point
3468 UsefulBits &= UsersUsefulBits;
3469}
3470
3471/// Create a machine node performing a notional SHL of Op by ShlAmount. If
3472/// ShlAmount is negative, do a (logical) right-shift instead. If ShlAmount is
3473/// 0, return Op unchanged.
3474static SDValue getLeftShift(SelectionDAG *CurDAG, SDValue Op, int ShlAmount) {
3475 if (ShlAmount == 0)
3476 return Op;
3477
3478 EVT VT = Op.getValueType();
3479 SDLoc dl(Op);
3480 unsigned BitWidth = VT.getSizeInBits();
3481 unsigned UBFMOpc = BitWidth == 32 ? AArch64::UBFMWri : AArch64::UBFMXri;
3482
3483 SDNode *ShiftNode;
3484 if (ShlAmount > 0) {
3485 // LSL wD, wN, #Amt == UBFM wD, wN, #32-Amt, #31-Amt
3486 ShiftNode = CurDAG->getMachineNode(
3487 UBFMOpc, dl, VT, Op,
3488 CurDAG->getTargetConstant(BitWidth - ShlAmount, dl, VT),
3489 CurDAG->getTargetConstant(BitWidth - 1 - ShlAmount, dl, VT));
3490 } else {
3491 // LSR wD, wN, #Amt == UBFM wD, wN, #Amt, #32-1
3492 assert(ShlAmount < 0 && "expected right shift");
3493 int ShrAmount = -ShlAmount;
3494 ShiftNode = CurDAG->getMachineNode(
3495 UBFMOpc, dl, VT, Op, CurDAG->getTargetConstant(ShrAmount, dl, VT),
3496 CurDAG->getTargetConstant(BitWidth - 1, dl, VT));
3497 }
3498
3499 return SDValue(ShiftNode, 0);
3500}
3501
3502// For bit-field-positioning pattern "(and (shl VAL, N), ShiftedMask)".
3503static bool isBitfieldPositioningOpFromAnd(SelectionDAG *CurDAG, SDValue Op,
3504 bool BiggerPattern,
3505 const uint64_t NonZeroBits,
3506 SDValue &Src, int &DstLSB,
3507 int &Width);
3508
3509// For bit-field-positioning pattern "shl VAL, N)".
3510static bool isBitfieldPositioningOpFromShl(SelectionDAG *CurDAG, SDValue Op,
3511 bool BiggerPattern,
3512 const uint64_t NonZeroBits,
3513 SDValue &Src, int &DstLSB,
3514 int &Width);
3515
3516/// Does this tree qualify as an attempt to move a bitfield into position,
3517/// essentially "(and (shl VAL, N), Mask)" or (shl VAL, N).
3519 bool BiggerPattern, SDValue &Src,
3520 int &DstLSB, int &Width) {
3521 EVT VT = Op.getValueType();
3522 unsigned BitWidth = VT.getSizeInBits();
3523 (void)BitWidth;
3524 assert(BitWidth == 32 || BitWidth == 64);
3525
3527
3528 // Non-zero in the sense that they're not provably zero, which is the key
3529 // point if we want to use this value
3530 const uint64_t NonZeroBits = (~Known.Zero).getZExtValue();
3531 if (!isShiftedMask_64(NonZeroBits))
3532 return false;
3533
3534 switch (Op.getOpcode()) {
3535 default:
3536 break;
3537 case ISD::AND:
3538 return isBitfieldPositioningOpFromAnd(CurDAG, Op, BiggerPattern,
3539 NonZeroBits, Src, DstLSB, Width);
3540 case ISD::SHL:
3541 return isBitfieldPositioningOpFromShl(CurDAG, Op, BiggerPattern,
3542 NonZeroBits, Src, DstLSB, Width);
3543 }
3544
3545 return false;
3546}
3547
3549 bool BiggerPattern,
3550 const uint64_t NonZeroBits,
3551 SDValue &Src, int &DstLSB,
3552 int &Width) {
3553 assert(isShiftedMask_64(NonZeroBits) && "Caller guaranteed");
3554
3555 EVT VT = Op.getValueType();
3556 assert((VT == MVT::i32 || VT == MVT::i64) &&
3557 "Caller guarantees VT is one of i32 or i64");
3558 (void)VT;
3559
3560 uint64_t AndImm;
3561 if (!isOpcWithIntImmediate(Op.getNode(), ISD::AND, AndImm))
3562 return false;
3563
3564 // If (~AndImm & NonZeroBits) is not zero at POS, we know that
3565 // 1) (AndImm & (1 << POS) == 0)
3566 // 2) the result of AND is not zero at POS bit (according to NonZeroBits)
3567 //
3568 // 1) and 2) don't agree so something must be wrong (e.g., in
3569 // 'SelectionDAG::computeKnownBits')
3570 assert((~AndImm & NonZeroBits) == 0 &&
3571 "Something must be wrong (e.g., in SelectionDAG::computeKnownBits)");
3572
3573 SDValue AndOp0 = Op.getOperand(0);
3574
3575 uint64_t ShlImm;
3576 SDValue ShlOp0;
3577 if (isOpcWithIntImmediate(AndOp0.getNode(), ISD::SHL, ShlImm)) {
3578 // For pattern "and(shl(val, N), shifted-mask)", 'ShlOp0' is set to 'val'.
3579 ShlOp0 = AndOp0.getOperand(0);
3580 } else if (VT == MVT::i64 && AndOp0.getOpcode() == ISD::ANY_EXTEND &&
3582 ShlImm)) {
3583 // For pattern "and(any_extend(shl(val, N)), shifted-mask)"
3584
3585 // ShlVal == shl(val, N), which is a left shift on a smaller type.
3586 SDValue ShlVal = AndOp0.getOperand(0);
3587
3588 // Since this is after type legalization and ShlVal is extended to MVT::i64,
3589 // expect VT to be MVT::i32.
3590 assert((ShlVal.getValueType() == MVT::i32) && "Expect VT to be MVT::i32.");
3591
3592 // Widens 'val' to MVT::i64 as the source of bit field positioning.
3593 ShlOp0 = Widen(CurDAG, ShlVal.getOperand(0));
3594 } else
3595 return false;
3596
3597 // For !BiggerPattern, bail out if the AndOp0 has more than one use, since
3598 // then we'll end up generating AndOp0+UBFIZ instead of just keeping
3599 // AndOp0+AND.
3600 if (!BiggerPattern && !AndOp0.hasOneUse())
3601 return false;
3602
3603 DstLSB = llvm::countr_zero(NonZeroBits);
3604 Width = llvm::countr_one(NonZeroBits >> DstLSB);
3605
3606 // Bail out on large Width. This happens when no proper combining / constant
3607 // folding was performed.
3608 if (Width >= (int)VT.getSizeInBits()) {
3609 // If VT is i64, Width > 64 is insensible since NonZeroBits is uint64_t, and
3610 // Width == 64 indicates a missed dag-combine from "(and val, AllOnes)" to
3611 // "val".
3612 // If VT is i32, what Width >= 32 means:
3613 // - For "(and (any_extend(shl val, N)), shifted-mask)", the`and` Op
3614 // demands at least 'Width' bits (after dag-combiner). This together with
3615 // `any_extend` Op (undefined higher bits) indicates missed combination
3616 // when lowering the 'and' IR instruction to an machine IR instruction.
3617 LLVM_DEBUG(
3618 dbgs()
3619 << "Found large Width in bit-field-positioning -- this indicates no "
3620 "proper combining / constant folding was performed\n");
3621 return false;
3622 }
3623
3624 // BFI encompasses sufficiently many nodes that it's worth inserting an extra
3625 // LSL/LSR if the mask in NonZeroBits doesn't quite match up with the ISD::SHL
3626 // amount. BiggerPattern is true when this pattern is being matched for BFI,
3627 // BiggerPattern is false when this pattern is being matched for UBFIZ, in
3628 // which case it is not profitable to insert an extra shift.
3629 if (ShlImm != uint64_t(DstLSB) && !BiggerPattern)
3630 return false;
3631
3632 Src = getLeftShift(CurDAG, ShlOp0, ShlImm - DstLSB);
3633 return true;
3634}
3635
3636// For node (shl (and val, mask), N)), returns true if the node is equivalent to
3637// UBFIZ.
3639 SDValue &Src, int &DstLSB,
3640 int &Width) {
3641 // Caller should have verified that N is a left shift with constant shift
3642 // amount; asserts that.
3643 assert(Op.getOpcode() == ISD::SHL &&
3644 "Op.getNode() should be a SHL node to call this function");
3645 assert(isIntImmediateEq(Op.getOperand(1), ShlImm) &&
3646 "Op.getNode() should shift ShlImm to call this function");
3647
3648 uint64_t AndImm = 0;
3649 SDValue Op0 = Op.getOperand(0);
3650 if (!isOpcWithIntImmediate(Op0.getNode(), ISD::AND, AndImm))
3651 return false;
3652
3653 const uint64_t ShiftedAndImm = ((AndImm << ShlImm) >> ShlImm);
3654 if (isMask_64(ShiftedAndImm)) {
3655 // AndImm is a superset of (AllOnes >> ShlImm); in other words, AndImm
3656 // should end with Mask, and could be prefixed with random bits if those
3657 // bits are shifted out.
3658 //
3659 // For example, xyz11111 (with {x,y,z} being 0 or 1) is fine if ShlImm >= 3;
3660 // the AND result corresponding to those bits are shifted out, so it's fine
3661 // to not extract them.
3662 Width = llvm::countr_one(ShiftedAndImm);
3663 DstLSB = ShlImm;
3664 Src = Op0.getOperand(0);
3665 return true;
3666 }
3667 return false;
3668}
3669
3671 bool BiggerPattern,
3672 const uint64_t NonZeroBits,
3673 SDValue &Src, int &DstLSB,
3674 int &Width) {
3675 assert(isShiftedMask_64(NonZeroBits) && "Caller guaranteed");
3676
3677 EVT VT = Op.getValueType();
3678 assert((VT == MVT::i32 || VT == MVT::i64) &&
3679 "Caller guarantees that type is i32 or i64");
3680 (void)VT;
3681
3682 uint64_t ShlImm;
3683 if (!isOpcWithIntImmediate(Op.getNode(), ISD::SHL, ShlImm))
3684 return false;
3685
3686 if (!BiggerPattern && !Op.hasOneUse())
3687 return false;
3688
3689 if (isSeveralBitsPositioningOpFromShl(ShlImm, Op, Src, DstLSB, Width))
3690 return true;
3691
3692 DstLSB = llvm::countr_zero(NonZeroBits);
3693 Width = llvm::countr_one(NonZeroBits >> DstLSB);
3694
3695 if (ShlImm != uint64_t(DstLSB) && !BiggerPattern)
3696 return false;
3697
3698 Src = getLeftShift(CurDAG, Op.getOperand(0), ShlImm - DstLSB);
3699 return true;
3700}
3701
3702static bool isShiftedMask(uint64_t Mask, EVT VT) {
3703 assert(VT == MVT::i32 || VT == MVT::i64);
3704 if (VT == MVT::i32)
3705 return isShiftedMask_32(Mask);
3706 return isShiftedMask_64(Mask);
3707}
3708
3709// Generate a BFI/BFXIL from 'or (and X, MaskImm), OrImm' iff the value being
3710// inserted only sets known zero bits.
3712 assert(N->getOpcode() == ISD::OR && "Expect a OR operation");
3713
3714 EVT VT = N->getValueType(0);
3715 if (VT != MVT::i32 && VT != MVT::i64)
3716 return false;
3717
3718 unsigned BitWidth = VT.getSizeInBits();
3719
3720 uint64_t OrImm;
3721 if (!isOpcWithIntImmediate(N, ISD::OR, OrImm))
3722 return false;
3723
3724 // Skip this transformation if the ORR immediate can be encoded in the ORR.
3725 // Otherwise, we'll trade an AND+ORR for ORR+BFI/BFXIL, which is most likely
3726 // performance neutral.
3728 return false;
3729
3730 uint64_t MaskImm;
3731 SDValue And = N->getOperand(0);
3732 // Must be a single use AND with an immediate operand.
3733 if (!And.hasOneUse() ||
3734 !isOpcWithIntImmediate(And.getNode(), ISD::AND, MaskImm))
3735 return false;
3736
3737 // Compute the Known Zero for the AND as this allows us to catch more general
3738 // cases than just looking for AND with imm.
3740
3741 // Non-zero in the sense that they're not provably zero, which is the key
3742 // point if we want to use this value.
3743 uint64_t NotKnownZero = (~Known.Zero).getZExtValue();
3744
3745 // The KnownZero mask must be a shifted mask (e.g., 1110..011, 11100..00).
3746 if (!isShiftedMask(Known.Zero.getZExtValue(), VT))
3747 return false;
3748
3749 // The bits being inserted must only set those bits that are known to be zero.
3750 if ((OrImm & NotKnownZero) != 0) {
3751 // FIXME: It's okay if the OrImm sets NotKnownZero bits to 1, but we don't
3752 // currently handle this case.
3753 return false;
3754 }
3755
3756 // BFI/BFXIL dst, src, #lsb, #width.
3757 int LSB = llvm::countr_one(NotKnownZero);
3758 int Width = BitWidth - APInt(BitWidth, NotKnownZero).popcount();
3759
3760 // BFI/BFXIL is an alias of BFM, so translate to BFM operands.
3761 unsigned ImmR = (BitWidth - LSB) % BitWidth;
3762 unsigned ImmS = Width - 1;
3763
3764 // If we're creating a BFI instruction avoid cases where we need more
3765 // instructions to materialize the BFI constant as compared to the original
3766 // ORR. A BFXIL will use the same constant as the original ORR, so the code
3767 // should be no worse in this case.
3768 bool IsBFI = LSB != 0;
3769 uint64_t BFIImm = OrImm >> LSB;
3770 if (IsBFI && !AArch64_AM::isLogicalImmediate(BFIImm, BitWidth)) {
3771 // We have a BFI instruction and we know the constant can't be materialized
3772 // with a ORR-immediate with the zero register.
3773 unsigned OrChunks = 0, BFIChunks = 0;
3774 for (unsigned Shift = 0; Shift < BitWidth; Shift += 16) {
3775 if (((OrImm >> Shift) & 0xFFFF) != 0)
3776 ++OrChunks;
3777 if (((BFIImm >> Shift) & 0xFFFF) != 0)
3778 ++BFIChunks;
3779 }
3780 if (BFIChunks > OrChunks)
3781 return false;
3782 }
3783
3784 // Materialize the constant to be inserted.
3785 SDLoc DL(N);
3786 unsigned MOVIOpc = VT == MVT::i32 ? AArch64::MOVi32imm : AArch64::MOVi64imm;
3787 SDNode *MOVI = CurDAG->getMachineNode(
3788 MOVIOpc, DL, VT, CurDAG->getTargetConstant(BFIImm, DL, VT));
3789
3790 // Create the BFI/BFXIL instruction.
3791 SDValue Ops[] = {And.getOperand(0), SDValue(MOVI, 0),
3792 CurDAG->getTargetConstant(ImmR, DL, VT),
3793 CurDAG->getTargetConstant(ImmS, DL, VT)};
3794 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
3795 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
3796 return true;
3797}
3798
3800 SDValue &ShiftedOperand,
3801 uint64_t &EncodedShiftImm) {
3802 // Avoid folding Dst into ORR-with-shift if Dst has other uses than ORR.
3803 if (!Dst.hasOneUse())
3804 return false;
3805
3806 EVT VT = Dst.getValueType();
3807 assert((VT == MVT::i32 || VT == MVT::i64) &&
3808 "Caller should guarantee that VT is one of i32 or i64");
3809 const unsigned SizeInBits = VT.getSizeInBits();
3810
3811 SDLoc DL(Dst.getNode());
3812 uint64_t AndImm, ShlImm;
3813 if (isOpcWithIntImmediate(Dst.getNode(), ISD::AND, AndImm) &&
3814 isShiftedMask_64(AndImm)) {
3815 // Avoid transforming 'DstOp0' if it has other uses than the AND node.
3816 SDValue DstOp0 = Dst.getOperand(0);
3817 if (!DstOp0.hasOneUse())
3818 return false;
3819
3820 // An example to illustrate the transformation
3821 // From:
3822 // lsr x8, x1, #1
3823 // and x8, x8, #0x3f80
3824 // bfxil x8, x1, #0, #7
3825 // To:
3826 // and x8, x23, #0x7f
3827 // ubfx x9, x23, #8, #7
3828 // orr x23, x8, x9, lsl #7
3829 //
3830 // The number of instructions remains the same, but ORR is faster than BFXIL
3831 // on many AArch64 processors (or as good as BFXIL if not faster). Besides,
3832 // the dependency chain is improved after the transformation.
3833 uint64_t SrlImm;
3834 if (isOpcWithIntImmediate(DstOp0.getNode(), ISD::SRL, SrlImm)) {
3835 uint64_t NumTrailingZeroInShiftedMask = llvm::countr_zero(AndImm);
3836 if ((SrlImm + NumTrailingZeroInShiftedMask) < SizeInBits) {
3837 unsigned MaskWidth =
3838 llvm::countr_one(AndImm >> NumTrailingZeroInShiftedMask);
3839 unsigned UBFMOpc =
3840 (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
3841 SDNode *UBFMNode = CurDAG->getMachineNode(
3842 UBFMOpc, DL, VT, DstOp0.getOperand(0),
3843 CurDAG->getTargetConstant(SrlImm + NumTrailingZeroInShiftedMask, DL,
3844 VT),
3845 CurDAG->getTargetConstant(
3846 SrlImm + NumTrailingZeroInShiftedMask + MaskWidth - 1, DL, VT));
3847 ShiftedOperand = SDValue(UBFMNode, 0);
3848 EncodedShiftImm = AArch64_AM::getShifterImm(
3849 AArch64_AM::LSL, NumTrailingZeroInShiftedMask);
3850 return true;
3851 }
3852 }
3853 return false;
3854 }
3855
3856 if (isOpcWithIntImmediate(Dst.getNode(), ISD::SHL, ShlImm)) {
3857 ShiftedOperand = Dst.getOperand(0);
3858 EncodedShiftImm = AArch64_AM::getShifterImm(AArch64_AM::LSL, ShlImm);
3859 return true;
3860 }
3861
3862 uint64_t SrlImm;
3863 if (isOpcWithIntImmediate(Dst.getNode(), ISD::SRL, SrlImm)) {
3864 ShiftedOperand = Dst.getOperand(0);
3865 EncodedShiftImm = AArch64_AM::getShifterImm(AArch64_AM::LSR, SrlImm);
3866 return true;
3867 }
3868 return false;
3869}
3870
3871// Given an 'ISD::OR' node that is going to be selected as BFM, analyze
3872// the operands and select it to AArch64::ORR with shifted registers if
3873// that's more efficient. Returns true iff selection to AArch64::ORR happens.
3874static bool tryOrrWithShift(SDNode *N, SDValue OrOpd0, SDValue OrOpd1,
3875 SDValue Src, SDValue Dst, SelectionDAG *CurDAG,
3876 const bool BiggerPattern) {
3877 EVT VT = N->getValueType(0);
3878 assert(N->getOpcode() == ISD::OR && "Expect N to be an OR node");
3879 assert(((N->getOperand(0) == OrOpd0 && N->getOperand(1) == OrOpd1) ||
3880 (N->getOperand(1) == OrOpd0 && N->getOperand(0) == OrOpd1)) &&
3881 "Expect OrOpd0 and OrOpd1 to be operands of ISD::OR");
3882 assert((VT == MVT::i32 || VT == MVT::i64) &&
3883 "Expect result type to be i32 or i64 since N is combinable to BFM");
3884 SDLoc DL(N);
3885
3886 // Bail out if BFM simplifies away one node in BFM Dst.
3887 if (OrOpd1 != Dst)
3888 return false;
3889
3890 const unsigned OrrOpc = (VT == MVT::i32) ? AArch64::ORRWrs : AArch64::ORRXrs;
3891 // For "BFM Rd, Rn, #immr, #imms", it's known that BFM simplifies away fewer
3892 // nodes from Rn (or inserts additional shift node) if BiggerPattern is true.
3893 if (BiggerPattern) {
3894 uint64_t SrcAndImm;
3895 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::AND, SrcAndImm) &&
3896 isMask_64(SrcAndImm) && OrOpd0.getOperand(0) == Src) {
3897 // OrOpd0 = AND Src, #Mask
3898 // So BFM simplifies away one AND node from Src and doesn't simplify away
3899 // nodes from Dst. If ORR with left-shifted operand also simplifies away
3900 // one node (from Rd), ORR is better since it has higher throughput and
3901 // smaller latency than BFM on many AArch64 processors (and for the rest
3902 // ORR is at least as good as BFM).
3903 SDValue ShiftedOperand;
3904 uint64_t EncodedShiftImm;
3905 if (isWorthFoldingIntoOrrWithShift(Dst, CurDAG, ShiftedOperand,
3906 EncodedShiftImm)) {
3907 SDValue Ops[] = {OrOpd0, ShiftedOperand,
3908 CurDAG->getTargetConstant(EncodedShiftImm, DL, VT)};
3909 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3910 return true;
3911 }
3912 }
3913 return false;
3914 }
3915
3916 assert((!BiggerPattern) && "BiggerPattern should be handled above");
3917
3918 uint64_t ShlImm;
3919 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::SHL, ShlImm)) {
3920 if (OrOpd0.getOperand(0) == Src && OrOpd0.hasOneUse()) {
3921 SDValue Ops[] = {
3922 Dst, Src,
3923 CurDAG->getTargetConstant(
3925 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3926 return true;
3927 }
3928
3929 // Select the following pattern to left-shifted operand rather than BFI.
3930 // %val1 = op ..
3931 // %val2 = shl %val1, #imm
3932 // %res = or %val1, %val2
3933 //
3934 // If N is selected to be BFI, we know that
3935 // 1) OrOpd0 would be the operand from which extract bits (i.e., folded into
3936 // BFI) 2) OrOpd1 would be the destination operand (i.e., preserved)
3937 //
3938 // Instead of selecting N to BFI, fold OrOpd0 as a left shift directly.
3939 if (OrOpd0.getOperand(0) == OrOpd1) {
3940 SDValue Ops[] = {
3941 OrOpd1, OrOpd1,
3942 CurDAG->getTargetConstant(
3944 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3945 return true;
3946 }
3947 }
3948
3949 uint64_t SrlImm;
3950 if (isOpcWithIntImmediate(OrOpd0.getNode(), ISD::SRL, SrlImm)) {
3951 // Select the following pattern to right-shifted operand rather than BFXIL.
3952 // %val1 = op ..
3953 // %val2 = lshr %val1, #imm
3954 // %res = or %val1, %val2
3955 //
3956 // If N is selected to be BFXIL, we know that
3957 // 1) OrOpd0 would be the operand from which extract bits (i.e., folded into
3958 // BFXIL) 2) OrOpd1 would be the destination operand (i.e., preserved)
3959 //
3960 // Instead of selecting N to BFXIL, fold OrOpd0 as a right shift directly.
3961 if (OrOpd0.getOperand(0) == OrOpd1) {
3962 SDValue Ops[] = {
3963 OrOpd1, OrOpd1,
3964 CurDAG->getTargetConstant(
3966 CurDAG->SelectNodeTo(N, OrrOpc, VT, Ops);
3967 return true;
3968 }
3969 }
3970
3971 return false;
3972}
3973
3974static bool tryBitfieldInsertOpFromOr(SDNode *N, const APInt &UsefulBits,
3975 SelectionDAG *CurDAG) {
3976 assert(N->getOpcode() == ISD::OR && "Expect a OR operation");
3977
3978 EVT VT = N->getValueType(0);
3979 if (VT != MVT::i32 && VT != MVT::i64)
3980 return false;
3981
3982 unsigned BitWidth = VT.getSizeInBits();
3983
3984 // Because of simplify-demanded-bits in DAGCombine, involved masks may not
3985 // have the expected shape. Try to undo that.
3986
3987 unsigned NumberOfIgnoredLowBits = UsefulBits.countr_zero();
3988 unsigned NumberOfIgnoredHighBits = UsefulBits.countl_zero();
3989
3990 // Given a OR operation, check if we have the following pattern
3991 // ubfm c, b, imm, imm2 (or something that does the same jobs, see
3992 // isBitfieldExtractOp)
3993 // d = e & mask2 ; where mask is a binary sequence of 1..10..0 and
3994 // countTrailingZeros(mask2) == imm2 - imm + 1
3995 // f = d | c
3996 // if yes, replace the OR instruction with:
3997 // f = BFM Opd0, Opd1, LSB, MSB ; where LSB = imm, and MSB = imm2
3998
3999 // OR is commutative, check all combinations of operand order and values of
4000 // BiggerPattern, i.e.
4001 // Opd0, Opd1, BiggerPattern=false
4002 // Opd1, Opd0, BiggerPattern=false
4003 // Opd0, Opd1, BiggerPattern=true
4004 // Opd1, Opd0, BiggerPattern=true
4005 // Several of these combinations may match, so check with BiggerPattern=false
4006 // first since that will produce better results by matching more instructions
4007 // and/or inserting fewer extra instructions.
4008 for (int I = 0; I < 4; ++I) {
4009
4010 SDValue Dst, Src;
4011 unsigned ImmR, ImmS;
4012 bool BiggerPattern = I / 2;
4013 SDValue OrOpd0Val = N->getOperand(I % 2);
4014 SDNode *OrOpd0 = OrOpd0Val.getNode();
4015 SDValue OrOpd1Val = N->getOperand((I + 1) % 2);
4016 SDNode *OrOpd1 = OrOpd1Val.getNode();
4017
4018 unsigned BFXOpc;
4019 int DstLSB, Width;
4020 if (isBitfieldExtractOp(CurDAG, OrOpd0, BFXOpc, Src, ImmR, ImmS,
4021 NumberOfIgnoredLowBits, BiggerPattern)) {
4022 // Check that the returned opcode is compatible with the pattern,
4023 // i.e., same type and zero extended (U and not S)
4024 if ((BFXOpc != AArch64::UBFMXri && VT == MVT::i64) ||
4025 (BFXOpc != AArch64::UBFMWri && VT == MVT::i32))
4026 continue;
4027
4028 // Compute the width of the bitfield insertion
4029 DstLSB = 0;
4030 Width = ImmS - ImmR + 1;
4031 // FIXME: This constraint is to catch bitfield insertion we may
4032 // want to widen the pattern if we want to grab general bitfield
4033 // move case
4034 if (Width <= 0)
4035 continue;
4036
4037 // If the mask on the insertee is correct, we have a BFXIL operation. We
4038 // can share the ImmR and ImmS values from the already-computed UBFM.
4039 } else if (isBitfieldPositioningOp(CurDAG, OrOpd0Val,
4040 BiggerPattern,
4041 Src, DstLSB, Width)) {
4042 ImmR = (BitWidth - DstLSB) % BitWidth;
4043 ImmS = Width - 1;
4044 } else
4045 continue;
4046
4047 // Check the second part of the pattern
4048 EVT VT = OrOpd1Val.getValueType();
4049 assert((VT == MVT::i32 || VT == MVT::i64) && "unexpected OR operand");
4050
4051 // Compute the Known Zero for the candidate of the first operand.
4052 // This allows to catch more general case than just looking for
4053 // AND with imm. Indeed, simplify-demanded-bits may have removed
4054 // the AND instruction because it proves it was useless.
4055 KnownBits Known = CurDAG->computeKnownBits(OrOpd1Val);
4056
4057 // Check if there is enough room for the second operand to appear
4058 // in the first one
4059 APInt BitsToBeInserted =
4060 APInt::getBitsSet(Known.getBitWidth(), DstLSB, DstLSB + Width);
4061
4062 if ((BitsToBeInserted & ~Known.Zero) != 0)
4063 continue;
4064
4065 // Set the first operand
4066 uint64_t Imm;
4067 if (isOpcWithIntImmediate(OrOpd1, ISD::AND, Imm) &&
4068 isBitfieldDstMask(Imm, BitsToBeInserted, NumberOfIgnoredHighBits, VT))
4069 // In that case, we can eliminate the AND
4070 Dst = OrOpd1->getOperand(0);
4071 else
4072 // Maybe the AND has been removed by simplify-demanded-bits
4073 // or is useful because it discards more bits
4074 Dst = OrOpd1Val;
4075
4076 // Before selecting ISD::OR node to AArch64::BFM, see if an AArch64::ORR
4077 // with shifted operand is more efficient.
4078 if (tryOrrWithShift(N, OrOpd0Val, OrOpd1Val, Src, Dst, CurDAG,
4079 BiggerPattern))
4080 return true;
4081
4082 // both parts match
4083 SDLoc DL(N);
4084 SDValue Ops[] = {Dst, Src, CurDAG->getTargetConstant(ImmR, DL, VT),
4085 CurDAG->getTargetConstant(ImmS, DL, VT)};
4086 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
4087 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4088 return true;
4089 }
4090
4091 // Generate a BFXIL from 'or (and X, Mask0Imm), (and Y, Mask1Imm)' iff
4092 // Mask0Imm and ~Mask1Imm are equivalent and one of the MaskImms is a shifted
4093 // mask (e.g., 0x000ffff0).
4094 uint64_t Mask0Imm, Mask1Imm;
4095 SDValue And0 = N->getOperand(0);
4096 SDValue And1 = N->getOperand(1);
4097 if (And0.hasOneUse() && And1.hasOneUse() &&
4098 isOpcWithIntImmediate(And0.getNode(), ISD::AND, Mask0Imm) &&
4099 isOpcWithIntImmediate(And1.getNode(), ISD::AND, Mask1Imm) &&
4100 APInt(BitWidth, Mask0Imm) == ~APInt(BitWidth, Mask1Imm) &&
4101 (isShiftedMask(Mask0Imm, VT) || isShiftedMask(Mask1Imm, VT))) {
4102
4103 // ORR is commutative, so canonicalize to the form 'or (and X, Mask0Imm),
4104 // (and Y, Mask1Imm)' where Mask1Imm is the shifted mask masking off the
4105 // bits to be inserted.
4106 if (isShiftedMask(Mask0Imm, VT)) {
4107 std::swap(And0, And1);
4108 std::swap(Mask0Imm, Mask1Imm);
4109 }
4110
4111 SDValue Src = And1->getOperand(0);
4112 SDValue Dst = And0->getOperand(0);
4113 unsigned LSB = llvm::countr_zero(Mask1Imm);
4114 int Width = BitWidth - APInt(BitWidth, Mask0Imm).popcount();
4115
4116 // The BFXIL inserts the low-order bits from a source register, so right
4117 // shift the needed bits into place.
4118 SDLoc DL(N);
4119 unsigned ShiftOpc = (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
4120 uint64_t LsrImm = LSB;
4121 if (Src->hasOneUse() &&
4122 isOpcWithIntImmediate(Src.getNode(), ISD::SRL, LsrImm) &&
4123 (LsrImm + LSB) < BitWidth) {
4124 Src = Src->getOperand(0);
4125 LsrImm += LSB;
4126 }
4127
4128 SDNode *LSR = CurDAG->getMachineNode(
4129 ShiftOpc, DL, VT, Src, CurDAG->getTargetConstant(LsrImm, DL, VT),
4130 CurDAG->getTargetConstant(BitWidth - 1, DL, VT));
4131
4132 // BFXIL is an alias of BFM, so translate to BFM operands.
4133 unsigned ImmR = (BitWidth - LSB) % BitWidth;
4134 unsigned ImmS = Width - 1;
4135
4136 // Create the BFXIL instruction.
4137 SDValue Ops[] = {Dst, SDValue(LSR, 0),
4138 CurDAG->getTargetConstant(ImmR, DL, VT),
4139 CurDAG->getTargetConstant(ImmS, DL, VT)};
4140 unsigned Opc = (VT == MVT::i32) ? AArch64::BFMWri : AArch64::BFMXri;
4141 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4142 return true;
4143 }
4144
4145 return false;
4146}
4147
4148bool AArch64DAGToDAGISel::tryBitfieldInsertOp(SDNode *N) {
4149 if (N->getOpcode() != ISD::OR)
4150 return false;
4151
4152 APInt NUsefulBits;
4153 getUsefulBits(SDValue(N, 0), NUsefulBits);
4154
4155 // If all bits are not useful, just return UNDEF.
4156 if (!NUsefulBits) {
4157 CurDAG->SelectNodeTo(N, TargetOpcode::IMPLICIT_DEF, N->getValueType(0));
4158 return true;
4159 }
4160
4161 if (tryBitfieldInsertOpFromOr(N, NUsefulBits, CurDAG))
4162 return true;
4163
4164 return tryBitfieldInsertOpFromOrAndImm(N, CurDAG);
4165}
4166
4167/// SelectBitfieldInsertInZeroOp - Match a UBFIZ instruction that is the
4168/// equivalent of a left shift by a constant amount followed by an and masking
4169/// out a contiguous set of bits.
4170bool AArch64DAGToDAGISel::tryBitfieldInsertInZeroOp(SDNode *N) {
4171 if (N->getOpcode() != ISD::AND)
4172 return false;
4173
4174 EVT VT = N->getValueType(0);
4175 if (VT != MVT::i32 && VT != MVT::i64)
4176 return false;
4177
4178 SDValue Op0;
4179 int DstLSB, Width;
4180 if (!isBitfieldPositioningOp(CurDAG, SDValue(N, 0), /*BiggerPattern=*/false,
4181 Op0, DstLSB, Width))
4182 return false;
4183
4184 // ImmR is the rotate right amount.
4185 unsigned ImmR = (VT.getSizeInBits() - DstLSB) % VT.getSizeInBits();
4186 // ImmS is the most significant bit of the source to be moved.
4187 unsigned ImmS = Width - 1;
4188
4189 SDLoc DL(N);
4190 SDValue Ops[] = {Op0, CurDAG->getTargetConstant(ImmR, DL, VT),
4191 CurDAG->getTargetConstant(ImmS, DL, VT)};
4192 unsigned Opc = (VT == MVT::i32) ? AArch64::UBFMWri : AArch64::UBFMXri;
4193 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4194 return true;
4195}
4196
4197/// tryShiftAmountMod - Take advantage of built-in mod of shift amount in
4198/// variable shift/rotate instructions.
4199bool AArch64DAGToDAGISel::tryShiftAmountMod(SDNode *N) {
4200 EVT VT = N->getValueType(0);
4201
4202 unsigned Opc;
4203 switch (N->getOpcode()) {
4204 case ISD::ROTR:
4205 Opc = (VT == MVT::i32) ? AArch64::RORVWr : AArch64::RORVXr;
4206 break;
4207 case ISD::SHL:
4208 Opc = (VT == MVT::i32) ? AArch64::LSLVWr : AArch64::LSLVXr;
4209 break;
4210 case ISD::SRL:
4211 Opc = (VT == MVT::i32) ? AArch64::LSRVWr : AArch64::LSRVXr;
4212 break;
4213 case ISD::SRA:
4214 Opc = (VT == MVT::i32) ? AArch64::ASRVWr : AArch64::ASRVXr;
4215 break;
4216 default:
4217 return false;
4218 }
4219
4220 uint64_t Size;
4221 uint64_t Bits;
4222 if (VT == MVT::i32) {
4223 Bits = 5;
4224 Size = 32;
4225 } else if (VT == MVT::i64) {
4226 Bits = 6;
4227 Size = 64;
4228 } else
4229 return false;
4230
4231 SDValue ShiftAmt = N->getOperand(1);
4232 SDLoc DL(N);
4233 SDValue NewShiftAmt;
4234
4235 // Skip over an extend of the shift amount.
4236 if (ShiftAmt->getOpcode() == ISD::ZERO_EXTEND ||
4237 ShiftAmt->getOpcode() == ISD::ANY_EXTEND)
4238 ShiftAmt = ShiftAmt->getOperand(0);
4239
4240 if (ShiftAmt->getOpcode() == ISD::ADD || ShiftAmt->getOpcode() == ISD::SUB) {
4241 SDValue Add0 = ShiftAmt->getOperand(0);
4242 SDValue Add1 = ShiftAmt->getOperand(1);
4243 uint64_t Add0Imm;
4244 uint64_t Add1Imm;
4245 if (isIntImmediate(Add1, Add1Imm) && (Add1Imm % Size == 0)) {
4246 // If we are shifting by X+/-N where N == 0 mod Size, then just shift by X
4247 // to avoid the ADD/SUB.
4248 NewShiftAmt = Add0;
4249 } else if (ShiftAmt->getOpcode() == ISD::SUB &&
4250 isIntImmediate(Add0, Add0Imm) && Add0Imm != 0 &&
4251 (Add0Imm % Size == 0)) {
4252 // If we are shifting by N-X where N == 0 mod Size, then just shift by -X
4253 // to generate a NEG instead of a SUB from a constant.
4254 unsigned NegOpc;
4255 unsigned ZeroReg;
4256 EVT SubVT = ShiftAmt->getValueType(0);
4257 if (SubVT == MVT::i32) {
4258 NegOpc = AArch64::SUBWrr;
4259 ZeroReg = AArch64::WZR;
4260 } else {
4261 assert(SubVT == MVT::i64);
4262 NegOpc = AArch64::SUBXrr;
4263 ZeroReg = AArch64::XZR;
4264 }
4265 SDValue Zero =
4266 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, SubVT);
4267 MachineSDNode *Neg =
4268 CurDAG->getMachineNode(NegOpc, DL, SubVT, Zero, Add1);
4269 NewShiftAmt = SDValue(Neg, 0);
4270 } else if (ShiftAmt->getOpcode() == ISD::SUB &&
4271 isIntImmediate(Add0, Add0Imm) && (Add0Imm % Size == Size - 1)) {
4272 // If we are shifting by N-X where N == -1 mod Size, then just shift by ~X
4273 // to generate a NOT instead of a SUB from a constant.
4274 unsigned NotOpc;
4275 unsigned ZeroReg;
4276 EVT SubVT = ShiftAmt->getValueType(0);
4277 if (SubVT == MVT::i32) {
4278 NotOpc = AArch64::ORNWrr;
4279 ZeroReg = AArch64::WZR;
4280 } else {
4281 assert(SubVT == MVT::i64);
4282 NotOpc = AArch64::ORNXrr;
4283 ZeroReg = AArch64::XZR;
4284 }
4285 SDValue Zero =
4286 CurDAG->getCopyFromReg(CurDAG->getEntryNode(), DL, ZeroReg, SubVT);
4287 MachineSDNode *Not =
4288 CurDAG->getMachineNode(NotOpc, DL, SubVT, Zero, Add1);
4289 NewShiftAmt = SDValue(Not, 0);
4290 } else
4291 return false;
4292 } else {
4293 // If the shift amount is masked with an AND, check that the mask covers the
4294 // bits that are implicitly ANDed off by the above opcodes and if so, skip
4295 // the AND.
4296 uint64_t MaskImm;
4297 if (!isOpcWithIntImmediate(ShiftAmt.getNode(), ISD::AND, MaskImm) &&
4298 !isOpcWithIntImmediate(ShiftAmt.getNode(), AArch64ISD::ANDS, MaskImm))
4299 return false;
4300
4301 if ((unsigned)llvm::countr_one(MaskImm) < Bits)
4302 return false;
4303
4304 NewShiftAmt = ShiftAmt->getOperand(0);
4305 }
4306
4307 // Narrow/widen the shift amount to match the size of the shift operation.
4308 if (VT == MVT::i32)
4309 NewShiftAmt = narrowIfNeeded(CurDAG, NewShiftAmt);
4310 else if (VT == MVT::i64 && NewShiftAmt->getValueType(0) == MVT::i32) {
4311 SDValue SubReg = CurDAG->getTargetConstant(AArch64::sub_32, DL, MVT::i32);
4312 MachineSDNode *Ext = CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, DL, VT,
4313 NewShiftAmt, SubReg);
4314 NewShiftAmt = SDValue(Ext, 0);
4315 }
4316
4317 SDValue Ops[] = {N->getOperand(0), NewShiftAmt};
4318 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
4319 return true;
4320}
4321
4323 SDValue &FixedPos,
4324 unsigned RegWidth,
4325 bool isReciprocal) {
4326 APFloat FVal(0.0);
4328 FVal = CN->getValueAPF();
4329 else if (LoadSDNode *LN = dyn_cast<LoadSDNode>(N)) {
4330 // Some otherwise illegal constants are allowed in this case.
4331 if (LN->getOperand(1).getOpcode() != AArch64ISD::ADDlow ||
4332 !isa<ConstantPoolSDNode>(LN->getOperand(1)->getOperand(1)))
4333 return false;
4334
4335 ConstantPoolSDNode *CN =
4336 dyn_cast<ConstantPoolSDNode>(LN->getOperand(1)->getOperand(1));
4337 FVal = cast<ConstantFP>(CN->getConstVal())->getValueAPF();
4338 } else
4339 return false;
4340
4341 if (unsigned FBits =
4342 CheckFixedPointOperandConstant(FVal, RegWidth, isReciprocal)) {
4343 FixedPos = CurDAG->getTargetConstant(FBits, SDLoc(N), MVT::i32);
4344 return true;
4345 }
4346
4347 return false;
4348}
4349
4351 SDValue N,
4352 SDValue &FixedPos,
4353 unsigned RegWidth,
4354 bool isReciprocal) {
4355 if ((N.getOpcode() == AArch64ISD::NVCAST || N.getOpcode() == ISD::BITCAST) &&
4356 N.getValueType().getScalarSizeInBits() ==
4357 N.getOperand(0).getValueType().getScalarSizeInBits())
4358 N = N.getOperand(0);
4359
4360 auto ImmToFloat = [RegWidth](APInt Imm) {
4361 switch (RegWidth) {
4362 case 16:
4363 return APFloat(APFloat::IEEEhalf(), Imm);
4364 case 32:
4365 return APFloat(APFloat::IEEEsingle(), Imm);
4366 case 64:
4367 return APFloat(APFloat::IEEEdouble(), Imm);
4368 default:
4369 llvm_unreachable("Unexpected RegWidth!");
4370 };
4371 };
4372
4373 APFloat FVal(0.0);
4374 switch (N->getOpcode()) {
4375 case AArch64ISD::MOVIshift:
4376 FVal = ImmToFloat(APInt(RegWidth, N.getConstantOperandVal(0)
4377 << N.getConstantOperandVal(1)));
4378 break;
4379 case AArch64ISD::FMOV:
4380 FVal = ImmToFloat(DecodeFMOVImm(N.getConstantOperandVal(0), RegWidth));
4381 break;
4382 case AArch64ISD::DUP:
4383 if (isa<ConstantSDNode>(N.getOperand(0)))
4384 FVal = ImmToFloat(N.getConstantOperandAPInt(0).trunc(RegWidth));
4385 else
4386 return false;
4387 break;
4388 default:
4389 return false;
4390 }
4391
4392 if (unsigned FBits =
4393 CheckFixedPointOperandConstant(FVal, RegWidth, isReciprocal)) {
4394 FixedPos = CurDAG->getTargetConstant(FBits, SDLoc(N), MVT::i32);
4395 return true;
4396 }
4397
4398 return false;
4399}
4400
4401bool AArch64DAGToDAGISel::SelectCVTFixedPosOperand(SDValue N, SDValue &FixedPos,
4402 unsigned RegWidth) {
4403 return checkCVTFixedPointOperandWithFBits(CurDAG, N, FixedPos, RegWidth,
4404 /*isReciprocal*/ false);
4405}
4406
4407bool AArch64DAGToDAGISel::SelectCVTFixedPointVec(SDValue N, SDValue &FixedPos,
4408 unsigned RegWidth) {
4410 CurDAG, N, FixedPos, RegWidth, /*isReciprocal*/ false);
4411}
4412
4413bool AArch64DAGToDAGISel::SelectCVTFixedPosRecipOperandVec(SDValue N,
4414 SDValue &FixedPos,
4415 unsigned RegWidth) {
4417 CurDAG, N, FixedPos, RegWidth, /*isReciprocal*/ true);
4418}
4419
4420bool AArch64DAGToDAGISel::SelectCVTFixedPosRecipOperand(SDValue N,
4421 SDValue &FixedPos,
4422 unsigned RegWidth) {
4423 return checkCVTFixedPointOperandWithFBits(CurDAG, N, FixedPos, RegWidth,
4424 /*isReciprocal*/ true);
4425}
4426
4427// Inspects a register string of the form o0:op1:CRn:CRm:op2 gets the fields
4428// of the string and obtains the integer values from them and combines these
4429// into a single value to be used in the MRS/MSR instruction.
4432 RegString.split(Fields, ':');
4433
4434 if (Fields.size() == 1)
4435 return -1;
4436
4437 assert(Fields.size() == 5
4438 && "Invalid number of fields in read register string");
4439
4441 bool AllIntFields = true;
4442
4443 for (StringRef Field : Fields) {
4444 unsigned IntField;
4445 AllIntFields &= !Field.getAsInteger(10, IntField);
4446 Ops.push_back(IntField);
4447 }
4448
4449 assert(AllIntFields &&
4450 "Unexpected non-integer value in special register string.");
4451 (void)AllIntFields;
4452
4453 // Need to combine the integer fields of the string into a single value
4454 // based on the bit encoding of MRS/MSR instruction.
4455 return (Ops[0] << 14) | (Ops[1] << 11) | (Ops[2] << 7) | (Ops[3] << 3) |
4456 (Ops[4]);
4457}
4458
4459// Lower the read_register intrinsic to an MRS instruction node if the special
4460// register string argument is either of the form detailed in the ALCE (the
4461// form described in getIntOperandsFromRegisterString) or is a named register
4462// known by the MRS SysReg mapper.
4463bool AArch64DAGToDAGISel::tryReadRegister(SDNode *N) {
4464 const auto *MD = cast<MDNodeSDNode>(N->getOperand(1));
4465 const auto *RegString = cast<MDString>(MD->getMD()->getOperand(0));
4466 SDLoc DL(N);
4467
4468 bool ReadIs128Bit = N->getOpcode() == AArch64ISD::MRRS;
4469
4470 unsigned Opcode64Bit = AArch64::MRS;
4471 int Imm = getIntOperandFromRegisterString(RegString->getString());
4472 if (Imm == -1) {
4473 // No match, Use the sysreg mapper to map the remaining possible strings to
4474 // the value for the register to be used for the instruction operand.
4475 const auto *TheReg =
4476 AArch64SysReg::lookupSysRegByName(RegString->getString());
4477 if (TheReg && TheReg->Readable &&
4478 TheReg->haveFeatures(Subtarget->getFeatureBits()))
4479 Imm = TheReg->Encoding;
4480 else
4481 Imm = AArch64SysReg::parseGenericRegister(RegString->getString());
4482
4483 if (Imm == -1) {
4484 // Still no match, see if this is "pc" or give up.
4485 if (!ReadIs128Bit && RegString->getString() == "pc") {
4486 Opcode64Bit = AArch64::ADR;
4487 Imm = 0;
4488 } else {
4489 // Not a system register. It may name an allocatable 64-bit GPR/FPR read
4490 // by the MSVC __getReg/__getRegFp intrinsics. Emit a pseudo that
4491 // carries the source register as an immediate so the read does not
4492 // reference an undefined physical register (which the machine verifier
4493 // rejects); the AsmPrinter materializes the real mov/fmov.
4494 Register PReg = Subtarget->getTargetLowering()->matchRegisterName(
4495 RegString->getString());
4496 unsigned PseudoOp = 0;
4497 if (AArch64::GPR64RegClass.contains(PReg))
4498 PseudoOp = AArch64::READ_REGISTER_GPR64;
4499 else if (AArch64::FPR64RegClass.contains(PReg))
4500 PseudoOp = AArch64::READ_REGISTER_FPR64;
4501 if (!ReadIs128Bit && PseudoOp && N->getValueType(0) == MVT::i64) {
4502 CurDAG->SelectNodeTo(N, PseudoOp, MVT::i64, MVT::Other,
4503 {CurDAG->getTargetConstant(PReg, DL, MVT::i32),
4504 N->getOperand(0)});
4505 return true;
4506 }
4507 return false;
4508 }
4509 }
4510 }
4511
4512 SDValue InChain = N->getOperand(0);
4513 SDValue SysRegImm = CurDAG->getTargetConstant(Imm, DL, MVT::i32);
4514 if (!ReadIs128Bit) {
4515 CurDAG->SelectNodeTo(N, Opcode64Bit, MVT::i64, MVT::Other /* Chain */,
4516 {SysRegImm, InChain});
4517 } else {
4518 SDNode *MRRS = CurDAG->getMachineNode(
4519 AArch64::MRRS, DL,
4520 {MVT::Untyped /* XSeqPair */, MVT::Other /* Chain */},
4521 {SysRegImm, InChain});
4522
4523 // Sysregs are not endian. The even register always contains the low half
4524 // of the register.
4525 SDValue Lo = CurDAG->getTargetExtractSubreg(AArch64::sube64, DL, MVT::i64,
4526 SDValue(MRRS, 0));
4527 SDValue Hi = CurDAG->getTargetExtractSubreg(AArch64::subo64, DL, MVT::i64,
4528 SDValue(MRRS, 0));
4529 SDValue OutChain = SDValue(MRRS, 1);
4530
4531 ReplaceUses(SDValue(N, 0), Lo);
4532 ReplaceUses(SDValue(N, 1), Hi);
4533 ReplaceUses(SDValue(N, 2), OutChain);
4534 };
4535 return true;
4536}
4537
4538// Lower the write_register intrinsic to an MSR instruction node if the special
4539// register string argument is either of the form detailed in the ALCE (the
4540// form described in getIntOperandsFromRegisterString) or is a named register
4541// known by the MSR SysReg mapper.
4542bool AArch64DAGToDAGISel::tryWriteRegister(SDNode *N) {
4543 const auto *MD = cast<MDNodeSDNode>(N->getOperand(1));
4544 const auto *RegString = cast<MDString>(MD->getMD()->getOperand(0));
4545 SDLoc DL(N);
4546
4547 bool WriteIs128Bit = N->getOpcode() == AArch64ISD::MSRR;
4548
4549 if (!WriteIs128Bit) {
4550 // Check if the register was one of those allowed as the pstatefield value
4551 // in the MSR (immediate) instruction. To accept the values allowed in the
4552 // pstatefield for the MSR (immediate) instruction, we also require that an
4553 // immediate value has been provided as an argument, we know that this is
4554 // the case as it has been ensured by semantic checking.
4555 auto trySelectPState = [&](auto PMapper, unsigned State) {
4556 if (PMapper) {
4557 assert(isa<ConstantSDNode>(N->getOperand(2)) &&
4558 "Expected a constant integer expression.");
4559 unsigned Reg = PMapper->Encoding;
4560 uint64_t Immed = N->getConstantOperandVal(2);
4561 CurDAG->SelectNodeTo(
4562 N, State, MVT::Other, CurDAG->getTargetConstant(Reg, DL, MVT::i32),
4563 CurDAG->getTargetConstant(Immed, DL, MVT::i16), N->getOperand(0));
4564 return true;
4565 }
4566 return false;
4567 };
4568
4569 if (trySelectPState(
4570 AArch64PState::lookupPStateImm0_15ByName(RegString->getString()),
4571 AArch64::MSRpstateImm4))
4572 return true;
4573 if (trySelectPState(
4574 AArch64PState::lookupPStateImm0_1ByName(RegString->getString()),
4575 AArch64::MSRpstateImm1))
4576 return true;
4577 }
4578
4579 int Imm = getIntOperandFromRegisterString(RegString->getString());
4580 if (Imm == -1) {
4581 // Use the sysreg mapper to attempt to map the remaining possible strings
4582 // to the value for the register to be used for the MSR (register)
4583 // instruction operand.
4584 auto TheReg = AArch64SysReg::lookupSysRegByName(RegString->getString());
4585 if (TheReg && TheReg->Writeable &&
4586 TheReg->haveFeatures(Subtarget->getFeatureBits()))
4587 Imm = TheReg->Encoding;
4588 else
4589 Imm = AArch64SysReg::parseGenericRegister(RegString->getString());
4590
4591 if (Imm == -1) {
4592 // Used by the MSVC __setReg/__setRegFp intrinsics. Copy the value into
4593 // the physical register and keep it live with a FAKE_USE so the write is
4594 // not dead-eliminated. (getRegisterByName rejects allocatable registers,
4595 // so the generic write path cannot handle these.)
4596 Register PReg = Subtarget->getTargetLowering()->matchRegisterName(
4597 RegString->getString());
4598 bool IsGPR = AArch64::GPR64RegClass.contains(PReg);
4599 bool IsFPR = AArch64::FPR64RegClass.contains(PReg);
4600 if (!WriteIs128Bit && (IsGPR || IsFPR) &&
4601 N->getOperand(2).getValueType() == MVT::i64) {
4602 SDValue Copy =
4603 CurDAG->getCopyToReg(N->getOperand(0), DL, PReg, N->getOperand(2));
4604 SDValue RegOp = CurDAG->getRegister(PReg, MVT::i64);
4605 SDNode *FakeUse = CurDAG->getMachineNode(TargetOpcode::FAKE_USE, DL,
4606 MVT::Other, {RegOp, Copy});
4607 ReplaceUses(SDValue(N, 0), SDValue(FakeUse, 0));
4608 CurDAG->RemoveDeadNode(N);
4609 return true;
4610 }
4611 return false;
4612 }
4613 }
4614
4615 SDValue InChain = N->getOperand(0);
4616 if (!WriteIs128Bit) {
4617 CurDAG->SelectNodeTo(N, AArch64::MSR, MVT::Other,
4618 CurDAG->getTargetConstant(Imm, DL, MVT::i32),
4619 N->getOperand(2), InChain);
4620 } else {
4621 // No endian swap. The lower half always goes into the even subreg, and the
4622 // higher half always into the odd supreg.
4623 SDNode *Pair = CurDAG->getMachineNode(
4624 TargetOpcode::REG_SEQUENCE, DL, MVT::Untyped /* XSeqPair */,
4625 {CurDAG->getTargetConstant(AArch64::XSeqPairsClassRegClass.getID(), DL,
4626 MVT::i32),
4627 N->getOperand(2),
4628 CurDAG->getTargetConstant(AArch64::sube64, DL, MVT::i32),
4629 N->getOperand(3),
4630 CurDAG->getTargetConstant(AArch64::subo64, DL, MVT::i32)});
4631
4632 CurDAG->SelectNodeTo(N, AArch64::MSRR, MVT::Other,
4633 CurDAG->getTargetConstant(Imm, DL, MVT::i32),
4634 SDValue(Pair, 0), InChain);
4635 }
4636
4637 return true;
4638}
4639
4640/// We've got special pseudo-instructions for these
4641bool AArch64DAGToDAGISel::SelectCMP_SWAP(SDNode *N) {
4642 unsigned Opcode;
4643 EVT MemTy = cast<MemSDNode>(N)->getMemoryVT();
4644
4645 // Leave IR for LSE if subtarget supports it.
4646 if (Subtarget->hasLSE()) return false;
4647
4648 if (MemTy == MVT::i8)
4649 Opcode = AArch64::CMP_SWAP_8;
4650 else if (MemTy == MVT::i16)
4651 Opcode = AArch64::CMP_SWAP_16;
4652 else if (MemTy == MVT::i32)
4653 Opcode = AArch64::CMP_SWAP_32;
4654 else if (MemTy == MVT::i64)
4655 Opcode = AArch64::CMP_SWAP_64;
4656 else
4657 llvm_unreachable("Unknown AtomicCmpSwap type");
4658
4659 MVT RegTy = MemTy == MVT::i64 ? MVT::i64 : MVT::i32;
4660 SDValue Ops[] = {N->getOperand(1), N->getOperand(2), N->getOperand(3),
4661 N->getOperand(0)};
4662 SDNode *CmpSwap = CurDAG->getMachineNode(
4663 Opcode, SDLoc(N),
4664 CurDAG->getVTList(RegTy, MVT::i32, MVT::Other), Ops);
4665
4666 MachineMemOperand *MemOp = cast<MemSDNode>(N)->getMemOperand();
4667 CurDAG->setNodeMemRefs(cast<MachineSDNode>(CmpSwap), {MemOp});
4668
4669 ReplaceUses(SDValue(N, 0), SDValue(CmpSwap, 0));
4670 ReplaceUses(SDValue(N, 1), SDValue(CmpSwap, 2));
4671 CurDAG->RemoveDeadNode(N);
4672
4673 return true;
4674}
4675
4677AArch64DAGToDAGISel::decodeMemoryHintFlags(MachineMemOperand *MMO) const {
4678 int MemoryHint = -1;
4679 const MDNode *MemCacheHint = MMO->getMemCacheHint();
4680 if (!MemCacheHint)
4681 return AArch64MemoryHint::NONE;
4682
4683 for (unsigned I = 0; I + 1 < MemCacheHint->getNumOperands(); I += 2) {
4684 if (MemCacheHint->getOperand(I).equalsStr("aarch64.mem_hint")) {
4685 const Metadata *Val = MemCacheHint->getOperand(I + 1).get();
4687 ->getZExtValue();
4688 }
4689 }
4690
4691 return toAArch64MemoryHint(MemoryHint);
4692}
4693
4694bool AArch64DAGToDAGISel::isAtomicMemoryHint(SDNode *N,
4695 AArch64MemoryHint Hint) const {
4696 return decodeMemoryHintFlags(cast<MemSDNode>(N)->getMemOperand()) == Hint;
4697}
4698
4699bool AArch64DAGToDAGISel::SelectSVEAddSubImm(SDValue N, MVT VT, SDValue &Imm,
4700 SDValue &Shift, bool Negate) {
4701 if (!isa<ConstantSDNode>(N))
4702 return false;
4703
4704 APInt Val =
4705 cast<ConstantSDNode>(N)->getAPIntValue().trunc(VT.getFixedSizeInBits());
4706
4707 return SelectSVEAddSubImm(SDLoc(N), Val, VT, Imm, Shift, Negate);
4708}
4709
4710bool AArch64DAGToDAGISel::SelectSVEAddSubImm(SDLoc DL, APInt Val, MVT VT,
4711 SDValue &Imm, SDValue &Shift,
4712 bool Negate) {
4713 if (Negate)
4714 Val = -Val;
4715
4716 switch (VT.SimpleTy) {
4717 case MVT::i8:
4718 // All immediates are supported.
4719 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4720 Imm = CurDAG->getTargetConstant(Val.getZExtValue(), DL, MVT::i32);
4721 return true;
4722 case MVT::i16:
4723 case MVT::i32:
4724 case MVT::i64:
4725 // Support 8bit unsigned immediates.
4726 if ((Val & ~0xff) == 0) {
4727 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4728 Imm = CurDAG->getTargetConstant(Val.getZExtValue(), DL, MVT::i32);
4729 return true;
4730 }
4731 // Support 16bit unsigned immediates that are a multiple of 256.
4732 if ((Val & ~0xff00) == 0) {
4733 Shift = CurDAG->getTargetConstant(8, DL, MVT::i32);
4734 Imm = CurDAG->getTargetConstant(Val.lshr(8).getZExtValue(), DL, MVT::i32);
4735 return true;
4736 }
4737 break;
4738 default:
4739 break;
4740 }
4741
4742 return false;
4743}
4744
4745bool AArch64DAGToDAGISel::SelectSVEAddSubSSatImm(SDValue N, MVT VT,
4746 SDValue &Imm, SDValue &Shift,
4747 bool Negate) {
4748 if (!isa<ConstantSDNode>(N))
4749 return false;
4750
4751 SDLoc DL(N);
4752 int64_t Val = cast<ConstantSDNode>(N)
4753 ->getAPIntValue()
4755 .getSExtValue();
4756
4757 if (Negate)
4758 Val = -Val;
4759
4760 // Signed saturating instructions treat their immediate operand as unsigned,
4761 // whereas the related intrinsics define their operands to be signed. This
4762 // means we can only use the immediate form when the operand is non-negative.
4763 if (Val < 0)
4764 return false;
4765
4766 switch (VT.SimpleTy) {
4767 case MVT::i8:
4768 // All positive immediates are supported.
4769 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4770 Imm = CurDAG->getTargetConstant(Val, DL, MVT::i32);
4771 return true;
4772 case MVT::i16:
4773 case MVT::i32:
4774 case MVT::i64:
4775 // Support 8bit positive immediates.
4776 if (Val <= 255) {
4777 Shift = CurDAG->getTargetConstant(0, DL, MVT::i32);
4778 Imm = CurDAG->getTargetConstant(Val, DL, MVT::i32);
4779 return true;
4780 }
4781 // Support 16bit positive immediates that are a multiple of 256.
4782 if (Val <= 65280 && Val % 256 == 0) {
4783 Shift = CurDAG->getTargetConstant(8, DL, MVT::i32);
4784 Imm = CurDAG->getTargetConstant(Val >> 8, DL, MVT::i32);
4785 return true;
4786 }
4787 break;
4788 default:
4789 break;
4790 }
4791
4792 return false;
4793}
4794
4795bool AArch64DAGToDAGISel::SelectSVECpyDupImm(SDValue N, MVT VT, SDValue &Imm,
4796 SDValue &Shift) {
4797 if (!isa<ConstantSDNode>(N))
4798 return false;
4799
4800 SDLoc DL(N);
4801 int64_t Val = cast<ConstantSDNode>(N)
4802 ->getAPIntValue()
4803 .trunc(VT.getFixedSizeInBits())
4804 .getSExtValue();
4805 int32_t ImmVal, ShiftVal;
4806 if (!AArch64_AM::isSVECpyDupImm(VT.getScalarSizeInBits(), Val, ImmVal,
4807 ShiftVal))
4808 return false;
4809
4810 Shift = CurDAG->getTargetConstant(ShiftVal, DL, MVT::i32);
4811 Imm = CurDAG->getTargetConstant(ImmVal, DL, MVT::i32);
4812 return true;
4813}
4814
4815bool AArch64DAGToDAGISel::SelectSVESignedArithImm(SDValue N, SDValue &Imm) {
4816 if (auto CNode = dyn_cast<ConstantSDNode>(N))
4817 return SelectSVESignedArithImm(SDLoc(N), CNode->getAPIntValue(), Imm);
4818 return false;
4819}
4820
4821bool AArch64DAGToDAGISel::SelectSVESignedArithImm(SDLoc DL, APInt Val,
4822 SDValue &Imm) {
4823 int64_t ImmVal = Val.getSExtValue();
4824 if (ImmVal >= -128 && ImmVal < 128) {
4825 Imm = CurDAG->getSignedTargetConstant(ImmVal, DL, MVT::i32);
4826 return true;
4827 }
4828 return false;
4829}
4830
4831bool AArch64DAGToDAGISel::SelectSVEArithImm(SDValue N, MVT VT, SDValue &Imm) {
4832 if (auto CNode = dyn_cast<ConstantSDNode>(N)) {
4833 uint64_t ImmVal = CNode->getZExtValue();
4834
4835 switch (VT.SimpleTy) {
4836 case MVT::i8:
4837 ImmVal &= 0xFF;
4838 break;
4839 case MVT::i16:
4840 ImmVal &= 0xFFFF;
4841 break;
4842 case MVT::i32:
4843 ImmVal &= 0xFFFFFFFF;
4844 break;
4845 case MVT::i64:
4846 break;
4847 default:
4848 llvm_unreachable("Unexpected type");
4849 }
4850
4851 if (ImmVal < 256) {
4852 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
4853 return true;
4854 }
4855 }
4856 return false;
4857}
4858
4859bool AArch64DAGToDAGISel::SelectSVELogicalImm(SDValue N, MVT VT, SDValue &Imm,
4860 bool Invert) {
4861 uint64_t ImmVal;
4862 if (auto CI = dyn_cast<ConstantSDNode>(N))
4863 ImmVal = CI->getZExtValue();
4864 else if (auto CFP = dyn_cast<ConstantFPSDNode>(N))
4865 ImmVal = CFP->getValueAPF().bitcastToAPInt().getZExtValue();
4866 else
4867 return false;
4868
4869 if (Invert)
4870 ImmVal = ~ImmVal;
4871
4872 uint64_t encoding;
4873 if (!AArch64_AM::isSVELogicalImm(VT.getScalarSizeInBits(), ImmVal, encoding))
4874 return false;
4875
4876 Imm = CurDAG->getTargetConstant(encoding, SDLoc(N), MVT::i64);
4877 return true;
4878}
4879
4880// SVE shift intrinsics allow shift amounts larger than the element's bitwidth.
4881// Rather than attempt to normalise everything we can sometimes saturate the
4882// shift amount during selection. This function also allows for consistent
4883// isel patterns by ensuring the resulting "Imm" node is of the i32 type
4884// required by the instructions.
4885bool AArch64DAGToDAGISel::SelectSVEShiftImm(SDValue N, uint64_t Low,
4886 uint64_t High, bool AllowSaturation,
4887 SDValue &Imm) {
4888 if (auto *CN = dyn_cast<ConstantSDNode>(N)) {
4889 uint64_t ImmVal = CN->getZExtValue();
4890
4891 // Reject shift amounts that are too small.
4892 if (ImmVal < Low)
4893 return false;
4894
4895 // Reject or saturate shift amounts that are too big.
4896 if (ImmVal > High) {
4897 if (!AllowSaturation)
4898 return false;
4899 ImmVal = High;
4900 }
4901
4902 Imm = CurDAG->getTargetConstant(ImmVal, SDLoc(N), MVT::i32);
4903 return true;
4904 }
4905
4906 return false;
4907}
4908
4909bool AArch64DAGToDAGISel::trySelectStackSlotTagP(SDNode *N) {
4910 // tagp(FrameIndex, IRGstack, tag_offset):
4911 // since the offset between FrameIndex and IRGstack is a compile-time
4912 // constant, this can be lowered to a single ADDG instruction.
4913 if (!(isa<FrameIndexSDNode>(N->getOperand(1)))) {
4914 return false;
4915 }
4916
4917 SDValue IRG_SP = N->getOperand(2);
4918 if (IRG_SP->getOpcode() != ISD::INTRINSIC_W_CHAIN ||
4919 IRG_SP->getConstantOperandVal(1) != Intrinsic::aarch64_irg_sp) {
4920 return false;
4921 }
4922
4923 const TargetLowering *TLI = getTargetLowering();
4924 SDLoc DL(N);
4925 int FI = cast<FrameIndexSDNode>(N->getOperand(1))->getIndex();
4926 SDValue FiOp = CurDAG->getTargetFrameIndex(
4927 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
4928 int TagOffset = N->getConstantOperandVal(3);
4929
4930 SDNode *Out = CurDAG->getMachineNode(
4931 AArch64::TAGPstack, DL, MVT::i64,
4932 {FiOp, CurDAG->getTargetConstant(0, DL, MVT::i64), N->getOperand(2),
4933 CurDAG->getTargetConstant(TagOffset, DL, MVT::i64)});
4934 ReplaceNode(N, Out);
4935 return true;
4936}
4937
4938void AArch64DAGToDAGISel::SelectTagP(SDNode *N) {
4939 assert(isa<ConstantSDNode>(N->getOperand(3)) &&
4940 "llvm.aarch64.tagp third argument must be an immediate");
4941 if (trySelectStackSlotTagP(N))
4942 return;
4943 // FIXME: above applies in any case when offset between Op1 and Op2 is a
4944 // compile-time constant, not just for stack allocations.
4945
4946 // General case for unrelated pointers in Op1 and Op2.
4947 SDLoc DL(N);
4948 int TagOffset = N->getConstantOperandVal(3);
4949 SDNode *N1 = CurDAG->getMachineNode(AArch64::SUBP, DL, MVT::i64,
4950 {N->getOperand(1), N->getOperand(2)});
4951 SDNode *N2 = CurDAG->getMachineNode(AArch64::ADDXrr, DL, MVT::i64,
4952 {SDValue(N1, 0), N->getOperand(2)});
4953 SDNode *N3 = CurDAG->getMachineNode(
4954 AArch64::ADDG, DL, MVT::i64,
4955 {SDValue(N2, 0), CurDAG->getTargetConstant(0, DL, MVT::i64),
4956 CurDAG->getTargetConstant(TagOffset, DL, MVT::i64)});
4957 ReplaceNode(N, N3);
4958}
4959
4960bool AArch64DAGToDAGISel::trySelectCastFixedLengthToScalableVector(SDNode *N) {
4961 assert(N->getOpcode() == ISD::INSERT_SUBVECTOR && "Invalid Node!");
4962
4963 // Bail when not a "cast" like insert_subvector.
4964 if (N->getConstantOperandVal(2) != 0)
4965 return false;
4966 if (!N->getOperand(0).isUndef())
4967 return false;
4968
4969 // Bail when normal isel should do the job.
4970 EVT VT = N->getValueType(0);
4971 EVT InVT = N->getOperand(1).getValueType();
4972 if (VT.isFixedLengthVector() || InVT.isScalableVector())
4973 return false;
4974 if (InVT.getSizeInBits() <= 128)
4975 return false;
4976
4977 // NOTE: We can only get here when doing fixed length SVE code generation.
4978 // We do manual selection because the types involved are not linked to real
4979 // registers (despite being legal) and must be coerced into SVE registers.
4980
4982 "Expected to insert into a packed scalable vector!");
4983
4984 SDLoc DL(N);
4985 auto RC = CurDAG->getTargetConstant(AArch64::ZPRRegClassID, DL, MVT::i64);
4986 ReplaceNode(N, CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS, DL, VT,
4987 N->getOperand(1), RC));
4988 return true;
4989}
4990
4991bool AArch64DAGToDAGISel::trySelectCastScalableToFixedLengthVector(SDNode *N) {
4992 assert(N->getOpcode() == ISD::EXTRACT_SUBVECTOR && "Invalid Node!");
4993
4994 // Bail when not a "cast" like extract_subvector.
4995 if (N->getConstantOperandVal(1) != 0)
4996 return false;
4997
4998 // Bail when normal isel can do the job.
4999 EVT VT = N->getValueType(0);
5000 EVT InVT = N->getOperand(0).getValueType();
5001 if (VT.isScalableVector() || InVT.isFixedLengthVector())
5002 return false;
5003 if (VT.getSizeInBits() <= 128)
5004 return false;
5005
5006 // NOTE: We can only get here when doing fixed length SVE code generation.
5007 // We do manual selection because the types involved are not linked to real
5008 // registers (despite being legal) and must be coerced into SVE registers.
5009
5011 "Expected to extract from a packed scalable vector!");
5012
5013 SDLoc DL(N);
5014 auto RC = CurDAG->getTargetConstant(AArch64::ZPRRegClassID, DL, MVT::i64);
5015 ReplaceNode(N, CurDAG->getMachineNode(TargetOpcode::COPY_TO_REGCLASS, DL, VT,
5016 N->getOperand(0), RC));
5017 return true;
5018}
5019
5020bool AArch64DAGToDAGISel::trySelectXAR(SDNode *N) {
5021 assert(N->getOpcode() == ISD::OR && "Expected OR instruction");
5022
5023 SDValue N0 = N->getOperand(0);
5024 SDValue N1 = N->getOperand(1);
5025
5026 EVT VT = N->getValueType(0);
5027 SDLoc DL(N);
5028
5029 // Essentially: rotr (xor(x, y), imm) -> xar (x, y, imm)
5030 // Rotate by a constant is a funnel shift in IR which is expanded to
5031 // an OR with shifted operands.
5032 // We do the following transform:
5033 // OR N0, N1 -> xar (x, y, imm)
5034 // Where:
5035 // N1 = SRL_PRED true, V, splat(imm) --> rotr amount
5036 // N0 = SHL_PRED true, V, splat(bits-imm)
5037 // V = (xor x, y)
5038 if (VT.isScalableVector() &&
5039 (Subtarget->hasSVE2() ||
5040 (Subtarget->hasSME() && Subtarget->isStreaming()))) {
5041 if (N0.getOpcode() != AArch64ISD::SHL_PRED ||
5042 N1.getOpcode() != AArch64ISD::SRL_PRED)
5043 std::swap(N0, N1);
5044 if (N0.getOpcode() != AArch64ISD::SHL_PRED ||
5045 N1.getOpcode() != AArch64ISD::SRL_PRED)
5046 return false;
5047
5048 auto *TLI = static_cast<const AArch64TargetLowering *>(getTargetLowering());
5049 if (!TLI->isAllActivePredicate(*CurDAG, N0.getOperand(0)) ||
5050 !TLI->isAllActivePredicate(*CurDAG, N1.getOperand(0)))
5051 return false;
5052
5053 if (N0.getOperand(1) != N1.getOperand(1))
5054 return false;
5055
5056 SDValue R1, R2;
5057 bool IsXOROperand = true;
5058 if (N0.getOperand(1).getOpcode() != ISD::XOR) {
5059 IsXOROperand = false;
5060 } else {
5061 R1 = N0.getOperand(1).getOperand(0);
5062 R2 = N1.getOperand(1).getOperand(1);
5063 }
5064
5065 APInt ShlAmt, ShrAmt;
5066 if (!ISD::isConstantSplatVector(N0.getOperand(2).getNode(), ShlAmt) ||
5068 return false;
5069
5070 if (ShlAmt + ShrAmt != VT.getScalarSizeInBits())
5071 return false;
5072
5073 if (!IsXOROperand) {
5074 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i64);
5075 SDNode *MOV = CurDAG->getMachineNode(AArch64::MOVIv2d_ns, DL, VT, Zero);
5076 SDValue MOVIV = SDValue(MOV, 0);
5077
5078 SDValue ZSub = CurDAG->getTargetConstant(AArch64::zsub, DL, MVT::i32);
5079 SDNode *SubRegToReg =
5080 CurDAG->getMachineNode(AArch64::SUBREG_TO_REG, DL, VT, MOVIV, ZSub);
5081
5082 R1 = N1->getOperand(1);
5083 R2 = SDValue(SubRegToReg, 0);
5084 }
5085
5086 SDValue Imm =
5087 CurDAG->getTargetConstant(ShrAmt.getZExtValue(), DL, MVT::i32);
5088
5089 SDValue Ops[] = {R1, R2, Imm};
5091 VT, {AArch64::XAR_ZZZI_B, AArch64::XAR_ZZZI_H, AArch64::XAR_ZZZI_S,
5092 AArch64::XAR_ZZZI_D})) {
5093 CurDAG->SelectNodeTo(N, Opc, VT, Ops);
5094 return true;
5095 }
5096 return false;
5097 }
5098
5099 // We have Neon SHA3 XAR operation for v2i64 but for types
5100 // v4i32, v8i16, v16i8 we can use SVE operations when SVE2-SHA3
5101 // is available.
5102 EVT SVT;
5103 switch (VT.getSimpleVT().SimpleTy) {
5104 case MVT::v4i32:
5105 case MVT::v2i32:
5106 SVT = MVT::nxv4i32;
5107 break;
5108 case MVT::v8i16:
5109 case MVT::v4i16:
5110 SVT = MVT::nxv8i16;
5111 break;
5112 case MVT::v16i8:
5113 case MVT::v8i8:
5114 SVT = MVT::nxv16i8;
5115 break;
5116 case MVT::v2i64:
5117 case MVT::v1i64:
5118 SVT = Subtarget->hasSHA3() ? MVT::v2i64 : MVT::nxv2i64;
5119 break;
5120 default:
5121 return false;
5122 }
5123
5124 if ((!SVT.isScalableVector() && !Subtarget->hasSHA3()) ||
5125 (SVT.isScalableVector() && !Subtarget->hasSVE2()))
5126 return false;
5127
5128 if (N0->getOpcode() != AArch64ISD::VSHL ||
5129 N1->getOpcode() != AArch64ISD::VLSHR)
5130 return false;
5131
5132 if (N0->getOperand(0) != N1->getOperand(0))
5133 return false;
5134
5135 SDValue R1, R2;
5136 bool IsXOROperand = true;
5137 if (N1->getOperand(0)->getOpcode() != ISD::XOR) {
5138 IsXOROperand = false;
5139 } else {
5140 SDValue XOR = N0.getOperand(0);
5141 R1 = XOR.getOperand(0);
5142 R2 = XOR.getOperand(1);
5143 }
5144
5145 unsigned HsAmt = N0.getConstantOperandVal(1);
5146 unsigned ShAmt = N1.getConstantOperandVal(1);
5147
5148 SDValue Imm = CurDAG->getTargetConstant(
5149 ShAmt, DL, N0.getOperand(1).getValueType(), false);
5150
5151 unsigned VTSizeInBits = VT.getScalarSizeInBits();
5152 if (ShAmt + HsAmt != VTSizeInBits)
5153 return false;
5154
5155 if (!IsXOROperand) {
5156 SDValue Zero = CurDAG->getTargetConstant(0, DL, MVT::i64);
5157 SDNode *MOV =
5158 CurDAG->getMachineNode(AArch64::MOVIv2d_ns, DL, MVT::v2i64, Zero);
5159 SDValue MOVIV = SDValue(MOV, 0);
5160
5161 R1 = N1->getOperand(0);
5162 R2 = MOVIV;
5163 }
5164
5165 if (SVT != VT) {
5166 SDValue Undef =
5167 SDValue(CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, SVT), 0);
5168
5169 if (SVT.isScalableVector() && VT.is64BitVector()) {
5170 EVT QVT = VT.getDoubleNumVectorElementsVT(*CurDAG->getContext());
5171
5172 SDValue UndefQ = SDValue(
5173 CurDAG->getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, QVT), 0);
5174 SDValue DSub = CurDAG->getTargetConstant(AArch64::dsub, DL, MVT::i32);
5175
5176 R1 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, QVT,
5177 UndefQ, R1, DSub),
5178 0);
5179 if (R2.getValueType() == VT)
5180 R2 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, QVT,
5181 UndefQ, R2, DSub),
5182 0);
5183 }
5184
5185 SDValue SubReg = CurDAG->getTargetConstant(
5186 (SVT.isScalableVector() ? AArch64::zsub : AArch64::dsub), DL, MVT::i32);
5187
5188 R1 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, SVT, Undef,
5189 R1, SubReg),
5190 0);
5191
5192 if (SVT.isScalableVector() || R2.getValueType() != SVT)
5193 R2 = SDValue(CurDAG->getMachineNode(AArch64::INSERT_SUBREG, DL, SVT,
5194 Undef, R2, SubReg),
5195 0);
5196 }
5197
5198 SDValue Ops[] = {R1, R2, Imm};
5199 SDNode *XAR = nullptr;
5200
5201 if (SVT.isScalableVector()) {
5203 SVT, {AArch64::XAR_ZZZI_B, AArch64::XAR_ZZZI_H, AArch64::XAR_ZZZI_S,
5204 AArch64::XAR_ZZZI_D}))
5205 XAR = CurDAG->getMachineNode(Opc, DL, SVT, Ops);
5206 } else {
5207 XAR = CurDAG->getMachineNode(AArch64::XAR, DL, SVT, Ops);
5208 }
5209
5210 assert(XAR && "Unexpected NULL value for XAR instruction in DAG");
5211
5212 if (SVT != VT) {
5213 if (VT.is64BitVector() && SVT.isScalableVector()) {
5214 EVT QVT = VT.getDoubleNumVectorElementsVT(*CurDAG->getContext());
5215
5216 SDValue ZSub = CurDAG->getTargetConstant(AArch64::zsub, DL, MVT::i32);
5217 SDNode *Q = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, QVT,
5218 SDValue(XAR, 0), ZSub);
5219
5220 SDValue DSub = CurDAG->getTargetConstant(AArch64::dsub, DL, MVT::i32);
5221 XAR = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, VT,
5222 SDValue(Q, 0), DSub);
5223 } else {
5224 SDValue SubReg = CurDAG->getTargetConstant(
5225 (SVT.isScalableVector() ? AArch64::zsub : AArch64::dsub), DL,
5226 MVT::i32);
5227 XAR = CurDAG->getMachineNode(AArch64::EXTRACT_SUBREG, DL, VT,
5228 SDValue(XAR, 0), SubReg);
5229 }
5230 }
5231 ReplaceNode(N, XAR);
5232 return true;
5233}
5234
5235/// Returns a copy from WZR or XZR. This can be used during instruction
5236/// selection (it does not require any further selection/legalization).
5238 assert(VT == MVT::i32 || VT == MVT::i64);
5239 return DAG.getCopyFromReg(DAG.getEntryNode(), DL,
5240 VT == MVT::i32 ? AArch64::WZR : AArch64::XZR, VT);
5241}
5242
5243void AArch64DAGToDAGISel::Select(SDNode *Node) {
5244 // If we have a custom node, we already have selected!
5245 if (Node->isMachineOpcode()) {
5246 LLVM_DEBUG(errs() << "== "; Node->dump(CurDAG); errs() << "\n");
5247 Node->setNodeId(-1);
5248 return;
5249 }
5250
5251 // Few custom selection stuff.
5252 EVT VT = Node->getValueType(0);
5253
5254 switch (Node->getOpcode()) {
5255 default:
5256 break;
5257
5259 if (SelectCMP_SWAP(Node))
5260 return;
5261 break;
5262
5263 case ISD::READ_REGISTER:
5264 case AArch64ISD::MRRS:
5265 if (tryReadRegister(Node))
5266 return;
5267 break;
5268
5270 case AArch64ISD::MSRR:
5271 if (tryWriteRegister(Node))
5272 return;
5273 break;
5274
5275 case ISD::LOAD: {
5276 // Try to select as an indexed load. Fall through to normal processing
5277 // if we can't.
5278 if (tryIndexedLoad(Node))
5279 return;
5280 break;
5281 }
5282
5283 case ISD::SRL:
5284 case ISD::AND:
5285 case ISD::SRA:
5287 if (tryBitfieldExtractOp(Node))
5288 return;
5289 if (tryBitfieldInsertInZeroOp(Node))
5290 return;
5291 [[fallthrough]];
5292 case ISD::ROTR:
5293 case ISD::SHL:
5294 if (tryShiftAmountMod(Node))
5295 return;
5296 break;
5297
5298 case ISD::OR:
5299 if (tryBitfieldInsertOp(Node))
5300 return;
5301 if (trySelectXAR(Node))
5302 return;
5303 break;
5304
5306 if (trySelectCastScalableToFixedLengthVector(Node))
5307 return;
5308 break;
5309 }
5310
5311 case ISD::INSERT_SUBVECTOR: {
5312 if (trySelectCastFixedLengthToScalableVector(Node))
5313 return;
5314 break;
5315 }
5316
5317 case AArch64ISD::CSEL:
5318 if (tryFoldCselToFMaxMin(Node))
5319 return;
5320 break;
5321
5322 case ISD::Constant: {
5323 // Materialize zero constants as copies from WZR/XZR. This allows
5324 // the coalescer to propagate these into other instructions.
5325 ConstantSDNode *ConstNode = cast<ConstantSDNode>(Node);
5326 if (ConstNode->isZero() && (VT == MVT::i32 || VT == MVT::i64)) {
5327 ReplaceNode(Node, getZeroRegister(*CurDAG, SDLoc(Node), VT).getNode());
5328 return;
5329 }
5330 break;
5331 }
5332
5333 case ISD::FrameIndex: {
5334 // Selects to ADDXri FI, 0 which in turn will become ADDXri SP, imm.
5335 int FI = cast<FrameIndexSDNode>(Node)->getIndex();
5336 unsigned Shifter = AArch64_AM::getShifterImm(AArch64_AM::LSL, 0);
5337 const TargetLowering *TLI = getTargetLowering();
5338 SDValue TFI = CurDAG->getTargetFrameIndex(
5339 FI, TLI->getPointerTy(CurDAG->getDataLayout()));
5340 SDLoc DL(Node);
5341 SDValue Ops[] = { TFI, CurDAG->getTargetConstant(0, DL, MVT::i32),
5342 CurDAG->getTargetConstant(Shifter, DL, MVT::i32) };
5343 CurDAG->SelectNodeTo(Node, AArch64::ADDXri, MVT::i64, Ops);
5344 return;
5345 }
5347 unsigned IntNo = Node->getConstantOperandVal(1);
5348 switch (IntNo) {
5349 default:
5350 break;
5351 case Intrinsic::aarch64_gcsss: {
5352 SDLoc DL(Node);
5353 SDValue Chain = Node->getOperand(0);
5354 SDValue Val = Node->getOperand(2);
5355 SDValue Zero = CurDAG->getCopyFromReg(Chain, DL, AArch64::XZR, MVT::i64);
5356 SDNode *SS1 =
5357 CurDAG->getMachineNode(AArch64::GCSSS1, DL, MVT::Other, Val, Chain);
5358 SDNode *SS2 = CurDAG->getMachineNode(AArch64::GCSSS2, DL, MVT::i64,
5359 MVT::Other, Zero, SDValue(SS1, 0));
5360 ReplaceNode(Node, SS2);
5361 return;
5362 }
5363 case Intrinsic::aarch64_ldaxp:
5364 case Intrinsic::aarch64_ldxp: {
5365 unsigned Op =
5366 IntNo == Intrinsic::aarch64_ldaxp ? AArch64::LDAXPX : AArch64::LDXPX;
5367 SDValue MemAddr = Node->getOperand(2);
5368 SDLoc DL(Node);
5369 SDValue Chain = Node->getOperand(0);
5370
5371 SDNode *Ld = CurDAG->getMachineNode(Op, DL, MVT::i64, MVT::i64,
5372 MVT::Other, MemAddr, Chain);
5373
5374 // Transfer memoperands.
5375 MachineMemOperand *MemOp =
5376 cast<MemIntrinsicSDNode>(Node)->getMemOperand();
5377 CurDAG->setNodeMemRefs(cast<MachineSDNode>(Ld), {MemOp});
5378 ReplaceNode(Node, Ld);
5379 return;
5380 }
5381 case Intrinsic::aarch64_stlxp:
5382 case Intrinsic::aarch64_stxp: {
5383 unsigned Op =
5384 IntNo == Intrinsic::aarch64_stlxp ? AArch64::STLXPX : AArch64::STXPX;
5385 SDLoc DL(Node);
5386 SDValue Chain = Node->getOperand(0);
5387 SDValue ValLo = Node->getOperand(2);
5388 SDValue ValHi = Node->getOperand(3);
5389 SDValue MemAddr = Node->getOperand(4);
5390
5391 // Place arguments in the right order.
5392 SDValue Ops[] = {ValLo, ValHi, MemAddr, Chain};
5393
5394 SDNode *St = CurDAG->getMachineNode(Op, DL, MVT::i32, MVT::Other, Ops);
5395 // Transfer memoperands.
5396 MachineMemOperand *MemOp =
5397 cast<MemIntrinsicSDNode>(Node)->getMemOperand();
5398 CurDAG->setNodeMemRefs(cast<MachineSDNode>(St), {MemOp});
5399
5400 ReplaceNode(Node, St);
5401 return;
5402 }
5403 case Intrinsic::aarch64_neon_ld1x2:
5404 if (VT == MVT::v8i8) {
5405 SelectLoad(Node, 2, AArch64::LD1Twov8b, AArch64::dsub0);
5406 return;
5407 } else if (VT == MVT::v16i8) {
5408 SelectLoad(Node, 2, AArch64::LD1Twov16b, AArch64::qsub0);
5409 return;
5410 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5411 SelectLoad(Node, 2, AArch64::LD1Twov4h, AArch64::dsub0);
5412 return;
5413 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5414 SelectLoad(Node, 2, AArch64::LD1Twov8h, AArch64::qsub0);
5415 return;
5416 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5417 SelectLoad(Node, 2, AArch64::LD1Twov2s, AArch64::dsub0);
5418 return;
5419 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5420 SelectLoad(Node, 2, AArch64::LD1Twov4s, AArch64::qsub0);
5421 return;
5422 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5423 SelectLoad(Node, 2, AArch64::LD1Twov1d, AArch64::dsub0);
5424 return;
5425 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5426 SelectLoad(Node, 2, AArch64::LD1Twov2d, AArch64::qsub0);
5427 return;
5428 }
5429 break;
5430 case Intrinsic::aarch64_neon_ld1x3:
5431 if (VT == MVT::v8i8) {
5432 SelectLoad(Node, 3, AArch64::LD1Threev8b, AArch64::dsub0);
5433 return;
5434 } else if (VT == MVT::v16i8) {
5435 SelectLoad(Node, 3, AArch64::LD1Threev16b, AArch64::qsub0);
5436 return;
5437 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5438 SelectLoad(Node, 3, AArch64::LD1Threev4h, AArch64::dsub0);
5439 return;
5440 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5441 SelectLoad(Node, 3, AArch64::LD1Threev8h, AArch64::qsub0);
5442 return;
5443 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5444 SelectLoad(Node, 3, AArch64::LD1Threev2s, AArch64::dsub0);
5445 return;
5446 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5447 SelectLoad(Node, 3, AArch64::LD1Threev4s, AArch64::qsub0);
5448 return;
5449 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5450 SelectLoad(Node, 3, AArch64::LD1Threev1d, AArch64::dsub0);
5451 return;
5452 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5453 SelectLoad(Node, 3, AArch64::LD1Threev2d, AArch64::qsub0);
5454 return;
5455 }
5456 break;
5457 case Intrinsic::aarch64_neon_ld1x4:
5458 if (VT == MVT::v8i8) {
5459 SelectLoad(Node, 4, AArch64::LD1Fourv8b, AArch64::dsub0);
5460 return;
5461 } else if (VT == MVT::v16i8) {
5462 SelectLoad(Node, 4, AArch64::LD1Fourv16b, AArch64::qsub0);
5463 return;
5464 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5465 SelectLoad(Node, 4, AArch64::LD1Fourv4h, AArch64::dsub0);
5466 return;
5467 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5468 SelectLoad(Node, 4, AArch64::LD1Fourv8h, AArch64::qsub0);
5469 return;
5470 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5471 SelectLoad(Node, 4, AArch64::LD1Fourv2s, AArch64::dsub0);
5472 return;
5473 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5474 SelectLoad(Node, 4, AArch64::LD1Fourv4s, AArch64::qsub0);
5475 return;
5476 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5477 SelectLoad(Node, 4, AArch64::LD1Fourv1d, AArch64::dsub0);
5478 return;
5479 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5480 SelectLoad(Node, 4, AArch64::LD1Fourv2d, AArch64::qsub0);
5481 return;
5482 }
5483 break;
5484 case Intrinsic::aarch64_neon_ld2:
5485 if (VT == MVT::v8i8) {
5486 SelectLoad(Node, 2, AArch64::LD2Twov8b, AArch64::dsub0);
5487 return;
5488 } else if (VT == MVT::v16i8) {
5489 SelectLoad(Node, 2, AArch64::LD2Twov16b, AArch64::qsub0);
5490 return;
5491 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5492 SelectLoad(Node, 2, AArch64::LD2Twov4h, AArch64::dsub0);
5493 return;
5494 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5495 SelectLoad(Node, 2, AArch64::LD2Twov8h, AArch64::qsub0);
5496 return;
5497 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5498 SelectLoad(Node, 2, AArch64::LD2Twov2s, AArch64::dsub0);
5499 return;
5500 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5501 SelectLoad(Node, 2, AArch64::LD2Twov4s, AArch64::qsub0);
5502 return;
5503 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5504 SelectLoad(Node, 2, AArch64::LD1Twov1d, AArch64::dsub0);
5505 return;
5506 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5507 SelectLoad(Node, 2, AArch64::LD2Twov2d, AArch64::qsub0);
5508 return;
5509 }
5510 break;
5511 case Intrinsic::aarch64_neon_ld3:
5512 if (VT == MVT::v8i8) {
5513 SelectLoad(Node, 3, AArch64::LD3Threev8b, AArch64::dsub0);
5514 return;
5515 } else if (VT == MVT::v16i8) {
5516 SelectLoad(Node, 3, AArch64::LD3Threev16b, AArch64::qsub0);
5517 return;
5518 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5519 SelectLoad(Node, 3, AArch64::LD3Threev4h, AArch64::dsub0);
5520 return;
5521 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5522 SelectLoad(Node, 3, AArch64::LD3Threev8h, AArch64::qsub0);
5523 return;
5524 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5525 SelectLoad(Node, 3, AArch64::LD3Threev2s, AArch64::dsub0);
5526 return;
5527 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5528 SelectLoad(Node, 3, AArch64::LD3Threev4s, AArch64::qsub0);
5529 return;
5530 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5531 SelectLoad(Node, 3, AArch64::LD1Threev1d, AArch64::dsub0);
5532 return;
5533 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5534 SelectLoad(Node, 3, AArch64::LD3Threev2d, AArch64::qsub0);
5535 return;
5536 }
5537 break;
5538 case Intrinsic::aarch64_neon_ld4:
5539 if (VT == MVT::v8i8) {
5540 SelectLoad(Node, 4, AArch64::LD4Fourv8b, AArch64::dsub0);
5541 return;
5542 } else if (VT == MVT::v16i8) {
5543 SelectLoad(Node, 4, AArch64::LD4Fourv16b, AArch64::qsub0);
5544 return;
5545 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5546 SelectLoad(Node, 4, AArch64::LD4Fourv4h, AArch64::dsub0);
5547 return;
5548 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5549 SelectLoad(Node, 4, AArch64::LD4Fourv8h, AArch64::qsub0);
5550 return;
5551 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5552 SelectLoad(Node, 4, AArch64::LD4Fourv2s, AArch64::dsub0);
5553 return;
5554 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5555 SelectLoad(Node, 4, AArch64::LD4Fourv4s, AArch64::qsub0);
5556 return;
5557 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5558 SelectLoad(Node, 4, AArch64::LD1Fourv1d, AArch64::dsub0);
5559 return;
5560 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5561 SelectLoad(Node, 4, AArch64::LD4Fourv2d, AArch64::qsub0);
5562 return;
5563 }
5564 break;
5565 case Intrinsic::aarch64_neon_ld2r:
5566 if (VT == MVT::v8i8) {
5567 SelectLoad(Node, 2, AArch64::LD2Rv8b, AArch64::dsub0);
5568 return;
5569 } else if (VT == MVT::v16i8) {
5570 SelectLoad(Node, 2, AArch64::LD2Rv16b, AArch64::qsub0);
5571 return;
5572 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5573 SelectLoad(Node, 2, AArch64::LD2Rv4h, AArch64::dsub0);
5574 return;
5575 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5576 SelectLoad(Node, 2, AArch64::LD2Rv8h, AArch64::qsub0);
5577 return;
5578 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5579 SelectLoad(Node, 2, AArch64::LD2Rv2s, AArch64::dsub0);
5580 return;
5581 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5582 SelectLoad(Node, 2, AArch64::LD2Rv4s, AArch64::qsub0);
5583 return;
5584 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5585 SelectLoad(Node, 2, AArch64::LD2Rv1d, AArch64::dsub0);
5586 return;
5587 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5588 SelectLoad(Node, 2, AArch64::LD2Rv2d, AArch64::qsub0);
5589 return;
5590 }
5591 break;
5592 case Intrinsic::aarch64_neon_ld3r:
5593 if (VT == MVT::v8i8) {
5594 SelectLoad(Node, 3, AArch64::LD3Rv8b, AArch64::dsub0);
5595 return;
5596 } else if (VT == MVT::v16i8) {
5597 SelectLoad(Node, 3, AArch64::LD3Rv16b, AArch64::qsub0);
5598 return;
5599 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5600 SelectLoad(Node, 3, AArch64::LD3Rv4h, AArch64::dsub0);
5601 return;
5602 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5603 SelectLoad(Node, 3, AArch64::LD3Rv8h, AArch64::qsub0);
5604 return;
5605 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5606 SelectLoad(Node, 3, AArch64::LD3Rv2s, AArch64::dsub0);
5607 return;
5608 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5609 SelectLoad(Node, 3, AArch64::LD3Rv4s, AArch64::qsub0);
5610 return;
5611 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5612 SelectLoad(Node, 3, AArch64::LD3Rv1d, AArch64::dsub0);
5613 return;
5614 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5615 SelectLoad(Node, 3, AArch64::LD3Rv2d, AArch64::qsub0);
5616 return;
5617 }
5618 break;
5619 case Intrinsic::aarch64_neon_ld4r:
5620 if (VT == MVT::v8i8) {
5621 SelectLoad(Node, 4, AArch64::LD4Rv8b, AArch64::dsub0);
5622 return;
5623 } else if (VT == MVT::v16i8) {
5624 SelectLoad(Node, 4, AArch64::LD4Rv16b, AArch64::qsub0);
5625 return;
5626 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
5627 SelectLoad(Node, 4, AArch64::LD4Rv4h, AArch64::dsub0);
5628 return;
5629 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
5630 SelectLoad(Node, 4, AArch64::LD4Rv8h, AArch64::qsub0);
5631 return;
5632 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
5633 SelectLoad(Node, 4, AArch64::LD4Rv2s, AArch64::dsub0);
5634 return;
5635 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
5636 SelectLoad(Node, 4, AArch64::LD4Rv4s, AArch64::qsub0);
5637 return;
5638 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
5639 SelectLoad(Node, 4, AArch64::LD4Rv1d, AArch64::dsub0);
5640 return;
5641 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
5642 SelectLoad(Node, 4, AArch64::LD4Rv2d, AArch64::qsub0);
5643 return;
5644 }
5645 break;
5646 case Intrinsic::aarch64_neon_ld2lane:
5647 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5648 SelectLoadLane(Node, 2, AArch64::LD2i8);
5649 return;
5650 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5651 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5652 SelectLoadLane(Node, 2, AArch64::LD2i16);
5653 return;
5654 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5655 VT == MVT::v2f32) {
5656 SelectLoadLane(Node, 2, AArch64::LD2i32);
5657 return;
5658 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5659 VT == MVT::v1f64) {
5660 SelectLoadLane(Node, 2, AArch64::LD2i64);
5661 return;
5662 }
5663 break;
5664 case Intrinsic::aarch64_neon_ld3lane:
5665 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5666 SelectLoadLane(Node, 3, AArch64::LD3i8);
5667 return;
5668 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5669 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5670 SelectLoadLane(Node, 3, AArch64::LD3i16);
5671 return;
5672 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5673 VT == MVT::v2f32) {
5674 SelectLoadLane(Node, 3, AArch64::LD3i32);
5675 return;
5676 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5677 VT == MVT::v1f64) {
5678 SelectLoadLane(Node, 3, AArch64::LD3i64);
5679 return;
5680 }
5681 break;
5682 case Intrinsic::aarch64_neon_ld4lane:
5683 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
5684 SelectLoadLane(Node, 4, AArch64::LD4i8);
5685 return;
5686 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
5687 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
5688 SelectLoadLane(Node, 4, AArch64::LD4i16);
5689 return;
5690 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
5691 VT == MVT::v2f32) {
5692 SelectLoadLane(Node, 4, AArch64::LD4i32);
5693 return;
5694 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
5695 VT == MVT::v1f64) {
5696 SelectLoadLane(Node, 4, AArch64::LD4i64);
5697 return;
5698 }
5699 break;
5700 case Intrinsic::aarch64_ld64b:
5701 SelectLoad(Node, 8, AArch64::LD64B, AArch64::x8sub_0);
5702 return;
5703 case Intrinsic::aarch64_sve_ld2q_sret: {
5704 SelectPredicatedLoad(Node, 2, 4, AArch64::LD2Q_IMM, AArch64::LD2Q, true);
5705 return;
5706 }
5707 case Intrinsic::aarch64_sve_ld3q_sret: {
5708 SelectPredicatedLoad(Node, 3, 4, AArch64::LD3Q_IMM, AArch64::LD3Q, true);
5709 return;
5710 }
5711 case Intrinsic::aarch64_sve_ld4q_sret: {
5712 SelectPredicatedLoad(Node, 4, 4, AArch64::LD4Q_IMM, AArch64::LD4Q, true);
5713 return;
5714 }
5715 case Intrinsic::aarch64_sve_ld2_sret: {
5716 if (VT == MVT::nxv16i8) {
5717 SelectPredicatedLoad(Node, 2, 0, AArch64::LD2B_IMM, AArch64::LD2B,
5718 true);
5719 return;
5720 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5721 VT == MVT::nxv8bf16) {
5722 SelectPredicatedLoad(Node, 2, 1, AArch64::LD2H_IMM, AArch64::LD2H,
5723 true);
5724 return;
5725 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5726 SelectPredicatedLoad(Node, 2, 2, AArch64::LD2W_IMM, AArch64::LD2W,
5727 true);
5728 return;
5729 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5730 SelectPredicatedLoad(Node, 2, 3, AArch64::LD2D_IMM, AArch64::LD2D,
5731 true);
5732 return;
5733 }
5734 break;
5735 }
5736 case Intrinsic::aarch64_sve_ld1_pn_x2: {
5737 if (VT == MVT::nxv16i8) {
5738 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5739 SelectContiguousMultiVectorLoad(
5740 Node, 2, 0, AArch64::LD1B_2Z_IMM_PSEUDO, AArch64::LD1B_2Z_PSEUDO);
5741 else if (Subtarget->hasSVE2p1())
5742 SelectContiguousMultiVectorLoad(Node, 2, 0, AArch64::LD1B_2Z_IMM,
5743 AArch64::LD1B_2Z);
5744 else
5745 break;
5746 return;
5747 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5748 VT == MVT::nxv8bf16) {
5749 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5750 SelectContiguousMultiVectorLoad(
5751 Node, 2, 1, AArch64::LD1H_2Z_IMM_PSEUDO, AArch64::LD1H_2Z_PSEUDO);
5752 else if (Subtarget->hasSVE2p1())
5753 SelectContiguousMultiVectorLoad(Node, 2, 1, AArch64::LD1H_2Z_IMM,
5754 AArch64::LD1H_2Z);
5755 else
5756 break;
5757 return;
5758 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5759 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5760 SelectContiguousMultiVectorLoad(
5761 Node, 2, 2, AArch64::LD1W_2Z_IMM_PSEUDO, AArch64::LD1W_2Z_PSEUDO);
5762 else if (Subtarget->hasSVE2p1())
5763 SelectContiguousMultiVectorLoad(Node, 2, 2, AArch64::LD1W_2Z_IMM,
5764 AArch64::LD1W_2Z);
5765 else
5766 break;
5767 return;
5768 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5769 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5770 SelectContiguousMultiVectorLoad(
5771 Node, 2, 3, AArch64::LD1D_2Z_IMM_PSEUDO, AArch64::LD1D_2Z_PSEUDO);
5772 else if (Subtarget->hasSVE2p1())
5773 SelectContiguousMultiVectorLoad(Node, 2, 3, AArch64::LD1D_2Z_IMM,
5774 AArch64::LD1D_2Z);
5775 else
5776 break;
5777 return;
5778 }
5779 break;
5780 }
5781 case Intrinsic::aarch64_sve_ld1_pn_x4: {
5782 if (VT == MVT::nxv16i8) {
5783 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5784 SelectContiguousMultiVectorLoad(
5785 Node, 4, 0, AArch64::LD1B_4Z_IMM_PSEUDO, AArch64::LD1B_4Z_PSEUDO);
5786 else if (Subtarget->hasSVE2p1())
5787 SelectContiguousMultiVectorLoad(Node, 4, 0, AArch64::LD1B_4Z_IMM,
5788 AArch64::LD1B_4Z);
5789 else
5790 break;
5791 return;
5792 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5793 VT == MVT::nxv8bf16) {
5794 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5795 SelectContiguousMultiVectorLoad(
5796 Node, 4, 1, AArch64::LD1H_4Z_IMM_PSEUDO, AArch64::LD1H_4Z_PSEUDO);
5797 else if (Subtarget->hasSVE2p1())
5798 SelectContiguousMultiVectorLoad(Node, 4, 1, AArch64::LD1H_4Z_IMM,
5799 AArch64::LD1H_4Z);
5800 else
5801 break;
5802 return;
5803 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5804 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5805 SelectContiguousMultiVectorLoad(
5806 Node, 4, 2, AArch64::LD1W_4Z_IMM_PSEUDO, AArch64::LD1W_4Z_PSEUDO);
5807 else if (Subtarget->hasSVE2p1())
5808 SelectContiguousMultiVectorLoad(Node, 4, 2, AArch64::LD1W_4Z_IMM,
5809 AArch64::LD1W_4Z);
5810 else
5811 break;
5812 return;
5813 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5814 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5815 SelectContiguousMultiVectorLoad(
5816 Node, 4, 3, AArch64::LD1D_4Z_IMM_PSEUDO, AArch64::LD1D_4Z_PSEUDO);
5817 else if (Subtarget->hasSVE2p1())
5818 SelectContiguousMultiVectorLoad(Node, 4, 3, AArch64::LD1D_4Z_IMM,
5819 AArch64::LD1D_4Z);
5820 else
5821 break;
5822 return;
5823 }
5824 break;
5825 }
5826 case Intrinsic::aarch64_sve_ldnt1_pn_x2: {
5827 if (VT == MVT::nxv16i8) {
5828 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5829 SelectContiguousMultiVectorLoad(Node, 2, 0,
5830 AArch64::LDNT1B_2Z_IMM_PSEUDO,
5831 AArch64::LDNT1B_2Z_PSEUDO);
5832 else if (Subtarget->hasSVE2p1())
5833 SelectContiguousMultiVectorLoad(Node, 2, 0, AArch64::LDNT1B_2Z_IMM,
5834 AArch64::LDNT1B_2Z);
5835 else
5836 break;
5837 return;
5838 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5839 VT == MVT::nxv8bf16) {
5840 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5841 SelectContiguousMultiVectorLoad(Node, 2, 1,
5842 AArch64::LDNT1H_2Z_IMM_PSEUDO,
5843 AArch64::LDNT1H_2Z_PSEUDO);
5844 else if (Subtarget->hasSVE2p1())
5845 SelectContiguousMultiVectorLoad(Node, 2, 1, AArch64::LDNT1H_2Z_IMM,
5846 AArch64::LDNT1H_2Z);
5847 else
5848 break;
5849 return;
5850 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5851 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5852 SelectContiguousMultiVectorLoad(Node, 2, 2,
5853 AArch64::LDNT1W_2Z_IMM_PSEUDO,
5854 AArch64::LDNT1W_2Z_PSEUDO);
5855 else if (Subtarget->hasSVE2p1())
5856 SelectContiguousMultiVectorLoad(Node, 2, 2, AArch64::LDNT1W_2Z_IMM,
5857 AArch64::LDNT1W_2Z);
5858 else
5859 break;
5860 return;
5861 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5862 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5863 SelectContiguousMultiVectorLoad(Node, 2, 3,
5864 AArch64::LDNT1D_2Z_IMM_PSEUDO,
5865 AArch64::LDNT1D_2Z_PSEUDO);
5866 else if (Subtarget->hasSVE2p1())
5867 SelectContiguousMultiVectorLoad(Node, 2, 3, AArch64::LDNT1D_2Z_IMM,
5868 AArch64::LDNT1D_2Z);
5869 else
5870 break;
5871 return;
5872 }
5873 break;
5874 }
5875 case Intrinsic::aarch64_sve_ldnt1_pn_x4: {
5876 if (VT == MVT::nxv16i8) {
5877 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5878 SelectContiguousMultiVectorLoad(Node, 4, 0,
5879 AArch64::LDNT1B_4Z_IMM_PSEUDO,
5880 AArch64::LDNT1B_4Z_PSEUDO);
5881 else if (Subtarget->hasSVE2p1())
5882 SelectContiguousMultiVectorLoad(Node, 4, 0, AArch64::LDNT1B_4Z_IMM,
5883 AArch64::LDNT1B_4Z);
5884 else
5885 break;
5886 return;
5887 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5888 VT == MVT::nxv8bf16) {
5889 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5890 SelectContiguousMultiVectorLoad(Node, 4, 1,
5891 AArch64::LDNT1H_4Z_IMM_PSEUDO,
5892 AArch64::LDNT1H_4Z_PSEUDO);
5893 else if (Subtarget->hasSVE2p1())
5894 SelectContiguousMultiVectorLoad(Node, 4, 1, AArch64::LDNT1H_4Z_IMM,
5895 AArch64::LDNT1H_4Z);
5896 else
5897 break;
5898 return;
5899 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5900 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5901 SelectContiguousMultiVectorLoad(Node, 4, 2,
5902 AArch64::LDNT1W_4Z_IMM_PSEUDO,
5903 AArch64::LDNT1W_4Z_PSEUDO);
5904 else if (Subtarget->hasSVE2p1())
5905 SelectContiguousMultiVectorLoad(Node, 4, 2, AArch64::LDNT1W_4Z_IMM,
5906 AArch64::LDNT1W_4Z);
5907 else
5908 break;
5909 return;
5910 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5911 if (Subtarget->hasSME2() && Subtarget->isStreaming())
5912 SelectContiguousMultiVectorLoad(Node, 4, 3,
5913 AArch64::LDNT1D_4Z_IMM_PSEUDO,
5914 AArch64::LDNT1D_4Z_PSEUDO);
5915 else if (Subtarget->hasSVE2p1())
5916 SelectContiguousMultiVectorLoad(Node, 4, 3, AArch64::LDNT1D_4Z_IMM,
5917 AArch64::LDNT1D_4Z);
5918 else
5919 break;
5920 return;
5921 }
5922 break;
5923 }
5924 case Intrinsic::aarch64_sve_ld3_sret: {
5925 if (VT == MVT::nxv16i8) {
5926 SelectPredicatedLoad(Node, 3, 0, AArch64::LD3B_IMM, AArch64::LD3B,
5927 true);
5928 return;
5929 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5930 VT == MVT::nxv8bf16) {
5931 SelectPredicatedLoad(Node, 3, 1, AArch64::LD3H_IMM, AArch64::LD3H,
5932 true);
5933 return;
5934 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5935 SelectPredicatedLoad(Node, 3, 2, AArch64::LD3W_IMM, AArch64::LD3W,
5936 true);
5937 return;
5938 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5939 SelectPredicatedLoad(Node, 3, 3, AArch64::LD3D_IMM, AArch64::LD3D,
5940 true);
5941 return;
5942 }
5943 break;
5944 }
5945 case Intrinsic::aarch64_sve_ld4_sret: {
5946 if (VT == MVT::nxv16i8) {
5947 SelectPredicatedLoad(Node, 4, 0, AArch64::LD4B_IMM, AArch64::LD4B,
5948 true);
5949 return;
5950 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5951 VT == MVT::nxv8bf16) {
5952 SelectPredicatedLoad(Node, 4, 1, AArch64::LD4H_IMM, AArch64::LD4H,
5953 true);
5954 return;
5955 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5956 SelectPredicatedLoad(Node, 4, 2, AArch64::LD4W_IMM, AArch64::LD4W,
5957 true);
5958 return;
5959 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5960 SelectPredicatedLoad(Node, 4, 3, AArch64::LD4D_IMM, AArch64::LD4D,
5961 true);
5962 return;
5963 }
5964 break;
5965 }
5966 case Intrinsic::aarch64_sme_read_hor_vg2: {
5967 if (VT == MVT::nxv16i8) {
5968 SelectMultiVectorMove<14, 2>(Node, 2, AArch64::ZAB0,
5969 AArch64::MOVA_2ZMXI_H_B);
5970 return;
5971 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5972 VT == MVT::nxv8bf16) {
5973 SelectMultiVectorMove<6, 2>(Node, 2, AArch64::ZAH0,
5974 AArch64::MOVA_2ZMXI_H_H);
5975 return;
5976 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5977 SelectMultiVectorMove<2, 2>(Node, 2, AArch64::ZAS0,
5978 AArch64::MOVA_2ZMXI_H_S);
5979 return;
5980 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
5981 SelectMultiVectorMove<0, 2>(Node, 2, AArch64::ZAD0,
5982 AArch64::MOVA_2ZMXI_H_D);
5983 return;
5984 }
5985 break;
5986 }
5987 case Intrinsic::aarch64_sme_read_ver_vg2: {
5988 if (VT == MVT::nxv16i8) {
5989 SelectMultiVectorMove<14, 2>(Node, 2, AArch64::ZAB0,
5990 AArch64::MOVA_2ZMXI_V_B);
5991 return;
5992 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
5993 VT == MVT::nxv8bf16) {
5994 SelectMultiVectorMove<6, 2>(Node, 2, AArch64::ZAH0,
5995 AArch64::MOVA_2ZMXI_V_H);
5996 return;
5997 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
5998 SelectMultiVectorMove<2, 2>(Node, 2, AArch64::ZAS0,
5999 AArch64::MOVA_2ZMXI_V_S);
6000 return;
6001 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6002 SelectMultiVectorMove<0, 2>(Node, 2, AArch64::ZAD0,
6003 AArch64::MOVA_2ZMXI_V_D);
6004 return;
6005 }
6006 break;
6007 }
6008 case Intrinsic::aarch64_sme_read_hor_vg4: {
6009 if (VT == MVT::nxv16i8) {
6010 SelectMultiVectorMove<12, 4>(Node, 4, AArch64::ZAB0,
6011 AArch64::MOVA_4ZMXI_H_B);
6012 return;
6013 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6014 VT == MVT::nxv8bf16) {
6015 SelectMultiVectorMove<4, 4>(Node, 4, AArch64::ZAH0,
6016 AArch64::MOVA_4ZMXI_H_H);
6017 return;
6018 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6019 SelectMultiVectorMove<0, 2>(Node, 4, AArch64::ZAS0,
6020 AArch64::MOVA_4ZMXI_H_S);
6021 return;
6022 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6023 SelectMultiVectorMove<0, 2>(Node, 4, AArch64::ZAD0,
6024 AArch64::MOVA_4ZMXI_H_D);
6025 return;
6026 }
6027 break;
6028 }
6029 case Intrinsic::aarch64_sme_read_ver_vg4: {
6030 if (VT == MVT::nxv16i8) {
6031 SelectMultiVectorMove<12, 4>(Node, 4, AArch64::ZAB0,
6032 AArch64::MOVA_4ZMXI_V_B);
6033 return;
6034 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6035 VT == MVT::nxv8bf16) {
6036 SelectMultiVectorMove<4, 4>(Node, 4, AArch64::ZAH0,
6037 AArch64::MOVA_4ZMXI_V_H);
6038 return;
6039 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6040 SelectMultiVectorMove<0, 4>(Node, 4, AArch64::ZAS0,
6041 AArch64::MOVA_4ZMXI_V_S);
6042 return;
6043 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6044 SelectMultiVectorMove<0, 4>(Node, 4, AArch64::ZAD0,
6045 AArch64::MOVA_4ZMXI_V_D);
6046 return;
6047 }
6048 break;
6049 }
6050 case Intrinsic::aarch64_sme_read_vg1x2: {
6051 SelectMultiVectorMove<7, 1>(Node, 2, AArch64::ZA,
6052 AArch64::MOVA_VG2_2ZMXI);
6053 return;
6054 }
6055 case Intrinsic::aarch64_sme_read_vg1x4: {
6056 SelectMultiVectorMove<7, 1>(Node, 4, AArch64::ZA,
6057 AArch64::MOVA_VG4_4ZMXI);
6058 return;
6059 }
6060 case Intrinsic::aarch64_sme_readz_horiz_x2: {
6061 if (VT == MVT::nxv16i8) {
6062 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_B_PSEUDO, 14, 2);
6063 return;
6064 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6065 VT == MVT::nxv8bf16) {
6066 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_H_PSEUDO, 6, 2);
6067 return;
6068 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6069 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_S_PSEUDO, 2, 2);
6070 return;
6071 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6072 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_H_D_PSEUDO, 0, 2);
6073 return;
6074 }
6075 break;
6076 }
6077 case Intrinsic::aarch64_sme_readz_vert_x2: {
6078 if (VT == MVT::nxv16i8) {
6079 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_B_PSEUDO, 14, 2);
6080 return;
6081 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6082 VT == MVT::nxv8bf16) {
6083 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_H_PSEUDO, 6, 2);
6084 return;
6085 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6086 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_S_PSEUDO, 2, 2);
6087 return;
6088 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6089 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_2ZMI_V_D_PSEUDO, 0, 2);
6090 return;
6091 }
6092 break;
6093 }
6094 case Intrinsic::aarch64_sme_readz_horiz_x4: {
6095 if (VT == MVT::nxv16i8) {
6096 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_B_PSEUDO, 12, 4);
6097 return;
6098 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6099 VT == MVT::nxv8bf16) {
6100 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_H_PSEUDO, 4, 4);
6101 return;
6102 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6103 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_S_PSEUDO, 0, 4);
6104 return;
6105 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6106 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_H_D_PSEUDO, 0, 4);
6107 return;
6108 }
6109 break;
6110 }
6111 case Intrinsic::aarch64_sme_readz_vert_x4: {
6112 if (VT == MVT::nxv16i8) {
6113 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_B_PSEUDO, 12, 4);
6114 return;
6115 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
6116 VT == MVT::nxv8bf16) {
6117 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_H_PSEUDO, 4, 4);
6118 return;
6119 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
6120 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_S_PSEUDO, 0, 4);
6121 return;
6122 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
6123 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_4ZMI_V_D_PSEUDO, 0, 4);
6124 return;
6125 }
6126 break;
6127 }
6128 case Intrinsic::aarch64_sme_readz_x2: {
6129 SelectMultiVectorMoveZ(Node, 2, AArch64::MOVAZ_VG2_2ZMXI_PSEUDO, 7, 1,
6130 AArch64::ZA);
6131 return;
6132 }
6133 case Intrinsic::aarch64_sme_readz_x4: {
6134 SelectMultiVectorMoveZ(Node, 4, AArch64::MOVAZ_VG4_4ZMXI_PSEUDO, 7, 1,
6135 AArch64::ZA);
6136 return;
6137 }
6138 case Intrinsic::swift_async_context_addr: {
6139 SDLoc DL(Node);
6140 SDValue Chain = Node->getOperand(0);
6141 SDValue CopyFP = CurDAG->getCopyFromReg(Chain, DL, AArch64::FP, MVT::i64);
6142 SDValue Res = SDValue(
6143 CurDAG->getMachineNode(AArch64::SUBXri, DL, MVT::i64, CopyFP,
6144 CurDAG->getTargetConstant(8, DL, MVT::i32),
6145 CurDAG->getTargetConstant(0, DL, MVT::i32)),
6146 0);
6147 ReplaceUses(SDValue(Node, 0), Res);
6148 ReplaceUses(SDValue(Node, 1), CopyFP.getValue(1));
6149 CurDAG->RemoveDeadNode(Node);
6150
6151 auto &MF = CurDAG->getMachineFunction();
6152 MF.getFrameInfo().setFrameAddressIsTaken(true);
6153 MF.getInfo<AArch64FunctionInfo>()->setHasSwiftAsyncContext(true);
6154 return;
6155 }
6156 case Intrinsic::aarch64_sme_luti2_lane_zt_x4: {
6158 Node->getValueType(0),
6159 {AArch64::LUTI2_4ZTZI_B, AArch64::LUTI2_4ZTZI_H,
6160 AArch64::LUTI2_4ZTZI_S}))
6161 // Second Immediate must be <= 3:
6162 SelectMultiVectorLutiLane(Node, 4, Opc, 3);
6163 return;
6164 }
6165 case Intrinsic::aarch64_sme_luti4_lane_zt_x4: {
6167 Node->getValueType(0),
6168 {0, AArch64::LUTI4_4ZTZI_H, AArch64::LUTI4_4ZTZI_S}))
6169 // Second Immediate must be <= 1:
6170 SelectMultiVectorLutiLane(Node, 4, Opc, 1);
6171 return;
6172 }
6173 case Intrinsic::aarch64_sme_luti2_lane_zt_x2: {
6175 Node->getValueType(0),
6176 {AArch64::LUTI2_2ZTZI_B, AArch64::LUTI2_2ZTZI_H,
6177 AArch64::LUTI2_2ZTZI_S}))
6178 // Second Immediate must be <= 7:
6179 SelectMultiVectorLutiLane(Node, 2, Opc, 7);
6180 return;
6181 }
6182 case Intrinsic::aarch64_sme_luti4_lane_zt_x2: {
6184 Node->getValueType(0),
6185 {AArch64::LUTI4_2ZTZI_B, AArch64::LUTI4_2ZTZI_H,
6186 AArch64::LUTI4_2ZTZI_S}))
6187 // Second Immediate must be <= 3:
6188 SelectMultiVectorLutiLane(Node, 2, Opc, 3);
6189 return;
6190 }
6191 case Intrinsic::aarch64_sme_luti4_zt_x4: {
6192 SelectMultiVectorLuti(Node, 4, AArch64::LUTI4_4ZZT2Z, 2);
6193 return;
6194 }
6195 case Intrinsic::aarch64_sme_luti6_zt_x4: {
6196 SelectMultiVectorLuti(Node, 4, AArch64::LUTI6_4ZT3Z, 3);
6197 return;
6198 }
6199 case Intrinsic::aarch64_sve_fp8_cvtl1_x2:
6201 Node->getValueType(0),
6202 {AArch64::BF1CVTL_2ZZ_BtoH, AArch64::F1CVTL_2ZZ_BtoH}))
6203 SelectCVTIntrinsicFP8(Node, 2, Opc);
6204 return;
6205 case Intrinsic::aarch64_sve_fp8_cvtl2_x2:
6207 Node->getValueType(0),
6208 {AArch64::BF2CVTL_2ZZ_BtoH, AArch64::F2CVTL_2ZZ_BtoH}))
6209 SelectCVTIntrinsicFP8(Node, 2, Opc);
6210 return;
6211 case Intrinsic::aarch64_sve_fp8_cvt1_x2:
6213 Node->getValueType(0),
6214 {AArch64::BF1CVT_2ZZ_BtoH, AArch64::F1CVT_2ZZ_BtoH}))
6215 SelectCVTIntrinsicFP8(Node, 2, Opc);
6216 return;
6217 case Intrinsic::aarch64_sve_fp8_cvt2_x2:
6219 Node->getValueType(0),
6220 {AArch64::BF2CVT_2ZZ_BtoH, AArch64::F2CVT_2ZZ_BtoH}))
6221 SelectCVTIntrinsicFP8(Node, 2, Opc);
6222 return;
6223 case Intrinsic::ptrauth_resign_load_relative:
6224 SelectPtrauthResign(Node);
6225 return;
6226 }
6227 } break;
6229 unsigned IntNo = Node->getConstantOperandVal(0);
6230 switch (IntNo) {
6231 default:
6232 break;
6233 case Intrinsic::aarch64_tagp:
6234 SelectTagP(Node);
6235 return;
6236
6237 case Intrinsic::ptrauth_auth:
6238 SelectPtrauthAuth(Node);
6239 return;
6240
6241 case Intrinsic::ptrauth_resign:
6242 SelectPtrauthResign(Node);
6243 return;
6244
6245 case Intrinsic::ptrauth_auth_with_pc_and_resign:
6246 SelectPtrauthResignWithPC(Node);
6247 return;
6248
6249 case Intrinsic::aarch64_neon_tbl2:
6250 SelectTable(Node, 2,
6251 VT == MVT::v8i8 ? AArch64::TBLv8i8Two : AArch64::TBLv16i8Two,
6252 false);
6253 return;
6254 case Intrinsic::aarch64_neon_tbl3:
6255 SelectTable(Node, 3, VT == MVT::v8i8 ? AArch64::TBLv8i8Three
6256 : AArch64::TBLv16i8Three,
6257 false);
6258 return;
6259 case Intrinsic::aarch64_neon_tbl4:
6260 SelectTable(Node, 4, VT == MVT::v8i8 ? AArch64::TBLv8i8Four
6261 : AArch64::TBLv16i8Four,
6262 false);
6263 return;
6264 case Intrinsic::aarch64_neon_tbx2:
6265 SelectTable(Node, 2,
6266 VT == MVT::v8i8 ? AArch64::TBXv8i8Two : AArch64::TBXv16i8Two,
6267 true);
6268 return;
6269 case Intrinsic::aarch64_neon_tbx3:
6270 SelectTable(Node, 3, VT == MVT::v8i8 ? AArch64::TBXv8i8Three
6271 : AArch64::TBXv16i8Three,
6272 true);
6273 return;
6274 case Intrinsic::aarch64_neon_tbx4:
6275 SelectTable(Node, 4, VT == MVT::v8i8 ? AArch64::TBXv8i8Four
6276 : AArch64::TBXv16i8Four,
6277 true);
6278 return;
6279 case Intrinsic::aarch64_sve_srshl_single_x2:
6281 Node->getValueType(0),
6282 {AArch64::SRSHL_VG2_2ZZ_B, AArch64::SRSHL_VG2_2ZZ_H,
6283 AArch64::SRSHL_VG2_2ZZ_S, AArch64::SRSHL_VG2_2ZZ_D}))
6284 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6285 return;
6286 case Intrinsic::aarch64_sve_srshl_single_x4:
6288 Node->getValueType(0),
6289 {AArch64::SRSHL_VG4_4ZZ_B, AArch64::SRSHL_VG4_4ZZ_H,
6290 AArch64::SRSHL_VG4_4ZZ_S, AArch64::SRSHL_VG4_4ZZ_D}))
6291 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6292 return;
6293 case Intrinsic::aarch64_sme_luti6_lane_x4_x2:
6294 SelectMultiVectorLuti6LaneX4(Node, 2);
6295 return;
6296 case Intrinsic::aarch64_sme_luti6_lane_x4_x3:
6297 SelectMultiVectorLuti6LaneX4(Node, 3);
6298 return;
6299 case Intrinsic::aarch64_sve_urshl_single_x2:
6301 Node->getValueType(0),
6302 {AArch64::URSHL_VG2_2ZZ_B, AArch64::URSHL_VG2_2ZZ_H,
6303 AArch64::URSHL_VG2_2ZZ_S, AArch64::URSHL_VG2_2ZZ_D}))
6304 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6305 return;
6306 case Intrinsic::aarch64_sve_urshl_single_x4:
6308 Node->getValueType(0),
6309 {AArch64::URSHL_VG4_4ZZ_B, AArch64::URSHL_VG4_4ZZ_H,
6310 AArch64::URSHL_VG4_4ZZ_S, AArch64::URSHL_VG4_4ZZ_D}))
6311 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6312 return;
6313 case Intrinsic::aarch64_sve_srshl_x2:
6315 Node->getValueType(0),
6316 {AArch64::SRSHL_VG2_2Z2Z_B, AArch64::SRSHL_VG2_2Z2Z_H,
6317 AArch64::SRSHL_VG2_2Z2Z_S, AArch64::SRSHL_VG2_2Z2Z_D}))
6318 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6319 return;
6320 case Intrinsic::aarch64_sve_srshl_x4:
6322 Node->getValueType(0),
6323 {AArch64::SRSHL_VG4_4Z4Z_B, AArch64::SRSHL_VG4_4Z4Z_H,
6324 AArch64::SRSHL_VG4_4Z4Z_S, AArch64::SRSHL_VG4_4Z4Z_D}))
6325 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6326 return;
6327 case Intrinsic::aarch64_sve_urshl_x2:
6329 Node->getValueType(0),
6330 {AArch64::URSHL_VG2_2Z2Z_B, AArch64::URSHL_VG2_2Z2Z_H,
6331 AArch64::URSHL_VG2_2Z2Z_S, AArch64::URSHL_VG2_2Z2Z_D}))
6332 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6333 return;
6334 case Intrinsic::aarch64_sve_urshl_x4:
6336 Node->getValueType(0),
6337 {AArch64::URSHL_VG4_4Z4Z_B, AArch64::URSHL_VG4_4Z4Z_H,
6338 AArch64::URSHL_VG4_4Z4Z_S, AArch64::URSHL_VG4_4Z4Z_D}))
6339 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6340 return;
6341 case Intrinsic::aarch64_sve_sqdmulh_single_vgx2:
6343 Node->getValueType(0),
6344 {AArch64::SQDMULH_VG2_2ZZ_B, AArch64::SQDMULH_VG2_2ZZ_H,
6345 AArch64::SQDMULH_VG2_2ZZ_S, AArch64::SQDMULH_VG2_2ZZ_D}))
6346 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6347 return;
6348 case Intrinsic::aarch64_sve_sqdmulh_single_vgx4:
6350 Node->getValueType(0),
6351 {AArch64::SQDMULH_VG4_4ZZ_B, AArch64::SQDMULH_VG4_4ZZ_H,
6352 AArch64::SQDMULH_VG4_4ZZ_S, AArch64::SQDMULH_VG4_4ZZ_D}))
6353 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6354 return;
6355 case Intrinsic::aarch64_sve_sqdmulh_vgx2:
6357 Node->getValueType(0),
6358 {AArch64::SQDMULH_VG2_2Z2Z_B, AArch64::SQDMULH_VG2_2Z2Z_H,
6359 AArch64::SQDMULH_VG2_2Z2Z_S, AArch64::SQDMULH_VG2_2Z2Z_D}))
6360 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6361 return;
6362 case Intrinsic::aarch64_sve_sqdmulh_vgx4:
6364 Node->getValueType(0),
6365 {AArch64::SQDMULH_VG4_4Z4Z_B, AArch64::SQDMULH_VG4_4Z4Z_H,
6366 AArch64::SQDMULH_VG4_4Z4Z_S, AArch64::SQDMULH_VG4_4Z4Z_D}))
6367 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6368 return;
6369 case Intrinsic::aarch64_sme_fp8_scale_single_x2:
6371 Node->getValueType(0),
6372 {0, AArch64::FSCALE_2ZZ_H, AArch64::FSCALE_2ZZ_S,
6373 AArch64::FSCALE_2ZZ_D}))
6374 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6375 return;
6376 case Intrinsic::aarch64_sme_fp8_scale_single_x4:
6378 Node->getValueType(0),
6379 {0, AArch64::FSCALE_4ZZ_H, AArch64::FSCALE_4ZZ_S,
6380 AArch64::FSCALE_4ZZ_D}))
6381 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6382 return;
6383 case Intrinsic::aarch64_sme_fp8_scale_x2:
6385 Node->getValueType(0),
6386 {0, AArch64::FSCALE_2Z2Z_H, AArch64::FSCALE_2Z2Z_S,
6387 AArch64::FSCALE_2Z2Z_D}))
6388 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6389 return;
6390 case Intrinsic::aarch64_sme_fp8_scale_x4:
6392 Node->getValueType(0),
6393 {0, AArch64::FSCALE_4Z4Z_H, AArch64::FSCALE_4Z4Z_S,
6394 AArch64::FSCALE_4Z4Z_D}))
6395 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6396 return;
6397 case Intrinsic::aarch64_sve_whilege_x2:
6399 Node->getValueType(0),
6400 {AArch64::WHILEGE_2PXX_B, AArch64::WHILEGE_2PXX_H,
6401 AArch64::WHILEGE_2PXX_S, AArch64::WHILEGE_2PXX_D}))
6402 SelectWhilePair(Node, Op);
6403 return;
6404 case Intrinsic::aarch64_sve_whilegt_x2:
6406 Node->getValueType(0),
6407 {AArch64::WHILEGT_2PXX_B, AArch64::WHILEGT_2PXX_H,
6408 AArch64::WHILEGT_2PXX_S, AArch64::WHILEGT_2PXX_D}))
6409 SelectWhilePair(Node, Op);
6410 return;
6411 case Intrinsic::aarch64_sve_whilehi_x2:
6413 Node->getValueType(0),
6414 {AArch64::WHILEHI_2PXX_B, AArch64::WHILEHI_2PXX_H,
6415 AArch64::WHILEHI_2PXX_S, AArch64::WHILEHI_2PXX_D}))
6416 SelectWhilePair(Node, Op);
6417 return;
6418 case Intrinsic::aarch64_sve_whilehs_x2:
6420 Node->getValueType(0),
6421 {AArch64::WHILEHS_2PXX_B, AArch64::WHILEHS_2PXX_H,
6422 AArch64::WHILEHS_2PXX_S, AArch64::WHILEHS_2PXX_D}))
6423 SelectWhilePair(Node, Op);
6424 return;
6425 case Intrinsic::aarch64_sve_whilele_x2:
6427 Node->getValueType(0),
6428 {AArch64::WHILELE_2PXX_B, AArch64::WHILELE_2PXX_H,
6429 AArch64::WHILELE_2PXX_S, AArch64::WHILELE_2PXX_D}))
6430 SelectWhilePair(Node, Op);
6431 return;
6432 case Intrinsic::aarch64_sve_whilelo_x2:
6434 Node->getValueType(0),
6435 {AArch64::WHILELO_2PXX_B, AArch64::WHILELO_2PXX_H,
6436 AArch64::WHILELO_2PXX_S, AArch64::WHILELO_2PXX_D}))
6437 SelectWhilePair(Node, Op);
6438 return;
6439 case Intrinsic::aarch64_sve_whilels_x2:
6441 Node->getValueType(0),
6442 {AArch64::WHILELS_2PXX_B, AArch64::WHILELS_2PXX_H,
6443 AArch64::WHILELS_2PXX_S, AArch64::WHILELS_2PXX_D}))
6444 SelectWhilePair(Node, Op);
6445 return;
6446 case Intrinsic::aarch64_sve_whilelt_x2:
6448 Node->getValueType(0),
6449 {AArch64::WHILELT_2PXX_B, AArch64::WHILELT_2PXX_H,
6450 AArch64::WHILELT_2PXX_S, AArch64::WHILELT_2PXX_D}))
6451 SelectWhilePair(Node, Op);
6452 return;
6453 case Intrinsic::aarch64_sve_smax_single_x2:
6455 Node->getValueType(0),
6456 {AArch64::SMAX_VG2_2ZZ_B, AArch64::SMAX_VG2_2ZZ_H,
6457 AArch64::SMAX_VG2_2ZZ_S, AArch64::SMAX_VG2_2ZZ_D}))
6458 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6459 return;
6460 case Intrinsic::aarch64_sve_umax_single_x2:
6462 Node->getValueType(0),
6463 {AArch64::UMAX_VG2_2ZZ_B, AArch64::UMAX_VG2_2ZZ_H,
6464 AArch64::UMAX_VG2_2ZZ_S, AArch64::UMAX_VG2_2ZZ_D}))
6465 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6466 return;
6467 case Intrinsic::aarch64_sve_fmax_single_x2:
6469 Node->getValueType(0),
6470 {AArch64::BFMAX_VG2_2ZZ_H, AArch64::FMAX_VG2_2ZZ_H,
6471 AArch64::FMAX_VG2_2ZZ_S, AArch64::FMAX_VG2_2ZZ_D}))
6472 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6473 return;
6474 case Intrinsic::aarch64_sve_smax_single_x4:
6476 Node->getValueType(0),
6477 {AArch64::SMAX_VG4_4ZZ_B, AArch64::SMAX_VG4_4ZZ_H,
6478 AArch64::SMAX_VG4_4ZZ_S, AArch64::SMAX_VG4_4ZZ_D}))
6479 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6480 return;
6481 case Intrinsic::aarch64_sve_umax_single_x4:
6483 Node->getValueType(0),
6484 {AArch64::UMAX_VG4_4ZZ_B, AArch64::UMAX_VG4_4ZZ_H,
6485 AArch64::UMAX_VG4_4ZZ_S, AArch64::UMAX_VG4_4ZZ_D}))
6486 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6487 return;
6488 case Intrinsic::aarch64_sve_fmax_single_x4:
6490 Node->getValueType(0),
6491 {AArch64::BFMAX_VG4_4ZZ_H, AArch64::FMAX_VG4_4ZZ_H,
6492 AArch64::FMAX_VG4_4ZZ_S, AArch64::FMAX_VG4_4ZZ_D}))
6493 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6494 return;
6495 case Intrinsic::aarch64_sve_smin_single_x2:
6497 Node->getValueType(0),
6498 {AArch64::SMIN_VG2_2ZZ_B, AArch64::SMIN_VG2_2ZZ_H,
6499 AArch64::SMIN_VG2_2ZZ_S, AArch64::SMIN_VG2_2ZZ_D}))
6500 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6501 return;
6502 case Intrinsic::aarch64_sve_umin_single_x2:
6504 Node->getValueType(0),
6505 {AArch64::UMIN_VG2_2ZZ_B, AArch64::UMIN_VG2_2ZZ_H,
6506 AArch64::UMIN_VG2_2ZZ_S, AArch64::UMIN_VG2_2ZZ_D}))
6507 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6508 return;
6509 case Intrinsic::aarch64_sve_fmin_single_x2:
6511 Node->getValueType(0),
6512 {AArch64::BFMIN_VG2_2ZZ_H, AArch64::FMIN_VG2_2ZZ_H,
6513 AArch64::FMIN_VG2_2ZZ_S, AArch64::FMIN_VG2_2ZZ_D}))
6514 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6515 return;
6516 case Intrinsic::aarch64_sve_smin_single_x4:
6518 Node->getValueType(0),
6519 {AArch64::SMIN_VG4_4ZZ_B, AArch64::SMIN_VG4_4ZZ_H,
6520 AArch64::SMIN_VG4_4ZZ_S, AArch64::SMIN_VG4_4ZZ_D}))
6521 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6522 return;
6523 case Intrinsic::aarch64_sve_umin_single_x4:
6525 Node->getValueType(0),
6526 {AArch64::UMIN_VG4_4ZZ_B, AArch64::UMIN_VG4_4ZZ_H,
6527 AArch64::UMIN_VG4_4ZZ_S, AArch64::UMIN_VG4_4ZZ_D}))
6528 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6529 return;
6530 case Intrinsic::aarch64_sve_fmin_single_x4:
6532 Node->getValueType(0),
6533 {AArch64::BFMIN_VG4_4ZZ_H, AArch64::FMIN_VG4_4ZZ_H,
6534 AArch64::FMIN_VG4_4ZZ_S, AArch64::FMIN_VG4_4ZZ_D}))
6535 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6536 return;
6537 case Intrinsic::aarch64_sve_smax_x2:
6539 Node->getValueType(0),
6540 {AArch64::SMAX_VG2_2Z2Z_B, AArch64::SMAX_VG2_2Z2Z_H,
6541 AArch64::SMAX_VG2_2Z2Z_S, AArch64::SMAX_VG2_2Z2Z_D}))
6542 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6543 return;
6544 case Intrinsic::aarch64_sve_umax_x2:
6546 Node->getValueType(0),
6547 {AArch64::UMAX_VG2_2Z2Z_B, AArch64::UMAX_VG2_2Z2Z_H,
6548 AArch64::UMAX_VG2_2Z2Z_S, AArch64::UMAX_VG2_2Z2Z_D}))
6549 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6550 return;
6551 case Intrinsic::aarch64_sve_fmax_x2:
6553 Node->getValueType(0),
6554 {AArch64::BFMAX_VG2_2Z2Z_H, AArch64::FMAX_VG2_2Z2Z_H,
6555 AArch64::FMAX_VG2_2Z2Z_S, AArch64::FMAX_VG2_2Z2Z_D}))
6556 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6557 return;
6558 case Intrinsic::aarch64_sve_smax_x4:
6560 Node->getValueType(0),
6561 {AArch64::SMAX_VG4_4Z4Z_B, AArch64::SMAX_VG4_4Z4Z_H,
6562 AArch64::SMAX_VG4_4Z4Z_S, AArch64::SMAX_VG4_4Z4Z_D}))
6563 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6564 return;
6565 case Intrinsic::aarch64_sve_umax_x4:
6567 Node->getValueType(0),
6568 {AArch64::UMAX_VG4_4Z4Z_B, AArch64::UMAX_VG4_4Z4Z_H,
6569 AArch64::UMAX_VG4_4Z4Z_S, AArch64::UMAX_VG4_4Z4Z_D}))
6570 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6571 return;
6572 case Intrinsic::aarch64_sve_fmax_x4:
6574 Node->getValueType(0),
6575 {AArch64::BFMAX_VG4_4Z2Z_H, AArch64::FMAX_VG4_4Z4Z_H,
6576 AArch64::FMAX_VG4_4Z4Z_S, AArch64::FMAX_VG4_4Z4Z_D}))
6577 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6578 return;
6579 case Intrinsic::aarch64_sme_famax_x2:
6581 Node->getValueType(0),
6582 {0, AArch64::FAMAX_2Z2Z_H, AArch64::FAMAX_2Z2Z_S,
6583 AArch64::FAMAX_2Z2Z_D}))
6584 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6585 return;
6586 case Intrinsic::aarch64_sme_famax_x4:
6588 Node->getValueType(0),
6589 {0, AArch64::FAMAX_4Z4Z_H, AArch64::FAMAX_4Z4Z_S,
6590 AArch64::FAMAX_4Z4Z_D}))
6591 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6592 return;
6593 case Intrinsic::aarch64_sme_famin_x2:
6595 Node->getValueType(0),
6596 {0, AArch64::FAMIN_2Z2Z_H, AArch64::FAMIN_2Z2Z_S,
6597 AArch64::FAMIN_2Z2Z_D}))
6598 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6599 return;
6600 case Intrinsic::aarch64_sme_famin_x4:
6602 Node->getValueType(0),
6603 {0, AArch64::FAMIN_4Z4Z_H, AArch64::FAMIN_4Z4Z_S,
6604 AArch64::FAMIN_4Z4Z_D}))
6605 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6606 return;
6607 case Intrinsic::aarch64_sve_smin_x2:
6609 Node->getValueType(0),
6610 {AArch64::SMIN_VG2_2Z2Z_B, AArch64::SMIN_VG2_2Z2Z_H,
6611 AArch64::SMIN_VG2_2Z2Z_S, AArch64::SMIN_VG2_2Z2Z_D}))
6612 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6613 return;
6614 case Intrinsic::aarch64_sve_umin_x2:
6616 Node->getValueType(0),
6617 {AArch64::UMIN_VG2_2Z2Z_B, AArch64::UMIN_VG2_2Z2Z_H,
6618 AArch64::UMIN_VG2_2Z2Z_S, AArch64::UMIN_VG2_2Z2Z_D}))
6619 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6620 return;
6621 case Intrinsic::aarch64_sve_fmin_x2:
6623 Node->getValueType(0),
6624 {AArch64::BFMIN_VG2_2Z2Z_H, AArch64::FMIN_VG2_2Z2Z_H,
6625 AArch64::FMIN_VG2_2Z2Z_S, AArch64::FMIN_VG2_2Z2Z_D}))
6626 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6627 return;
6628 case Intrinsic::aarch64_sve_smin_x4:
6630 Node->getValueType(0),
6631 {AArch64::SMIN_VG4_4Z4Z_B, AArch64::SMIN_VG4_4Z4Z_H,
6632 AArch64::SMIN_VG4_4Z4Z_S, AArch64::SMIN_VG4_4Z4Z_D}))
6633 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6634 return;
6635 case Intrinsic::aarch64_sve_umin_x4:
6637 Node->getValueType(0),
6638 {AArch64::UMIN_VG4_4Z4Z_B, AArch64::UMIN_VG4_4Z4Z_H,
6639 AArch64::UMIN_VG4_4Z4Z_S, AArch64::UMIN_VG4_4Z4Z_D}))
6640 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6641 return;
6642 case Intrinsic::aarch64_sve_fmin_x4:
6644 Node->getValueType(0),
6645 {AArch64::BFMIN_VG4_4Z2Z_H, AArch64::FMIN_VG4_4Z4Z_H,
6646 AArch64::FMIN_VG4_4Z4Z_S, AArch64::FMIN_VG4_4Z4Z_D}))
6647 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6648 return;
6649 case Intrinsic::aarch64_sve_fmaxnm_single_x2 :
6651 Node->getValueType(0),
6652 {AArch64::BFMAXNM_VG2_2ZZ_H, AArch64::FMAXNM_VG2_2ZZ_H,
6653 AArch64::FMAXNM_VG2_2ZZ_S, AArch64::FMAXNM_VG2_2ZZ_D}))
6654 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6655 return;
6656 case Intrinsic::aarch64_sve_fmaxnm_single_x4 :
6658 Node->getValueType(0),
6659 {AArch64::BFMAXNM_VG4_4ZZ_H, AArch64::FMAXNM_VG4_4ZZ_H,
6660 AArch64::FMAXNM_VG4_4ZZ_S, AArch64::FMAXNM_VG4_4ZZ_D}))
6661 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6662 return;
6663 case Intrinsic::aarch64_sve_fminnm_single_x2:
6665 Node->getValueType(0),
6666 {AArch64::BFMINNM_VG2_2ZZ_H, AArch64::FMINNM_VG2_2ZZ_H,
6667 AArch64::FMINNM_VG2_2ZZ_S, AArch64::FMINNM_VG2_2ZZ_D}))
6668 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6669 return;
6670 case Intrinsic::aarch64_sve_fminnm_single_x4:
6672 Node->getValueType(0),
6673 {AArch64::BFMINNM_VG4_4ZZ_H, AArch64::FMINNM_VG4_4ZZ_H,
6674 AArch64::FMINNM_VG4_4ZZ_S, AArch64::FMINNM_VG4_4ZZ_D}))
6675 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6676 return;
6677 case Intrinsic::aarch64_sve_fscale_single_x4:
6678 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::BFSCALE_4ZZ);
6679 return;
6680 case Intrinsic::aarch64_sve_fscale_single_x2:
6681 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::BFSCALE_2ZZ);
6682 return;
6683 case Intrinsic::aarch64_sve_fmul_single_x4:
6685 Node->getValueType(0),
6686 {AArch64::BFMUL_4ZZ, AArch64::FMUL_4ZZ_H, AArch64::FMUL_4ZZ_S,
6687 AArch64::FMUL_4ZZ_D}))
6688 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6689 return;
6690 case Intrinsic::aarch64_sve_fmul_single_x2:
6692 Node->getValueType(0),
6693 {AArch64::BFMUL_2ZZ, AArch64::FMUL_2ZZ_H, AArch64::FMUL_2ZZ_S,
6694 AArch64::FMUL_2ZZ_D}))
6695 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6696 return;
6697 case Intrinsic::aarch64_sve_fmaxnm_x2:
6699 Node->getValueType(0),
6700 {AArch64::BFMAXNM_VG2_2Z2Z_H, AArch64::FMAXNM_VG2_2Z2Z_H,
6701 AArch64::FMAXNM_VG2_2Z2Z_S, AArch64::FMAXNM_VG2_2Z2Z_D}))
6702 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6703 return;
6704 case Intrinsic::aarch64_sve_fmaxnm_x4:
6706 Node->getValueType(0),
6707 {AArch64::BFMAXNM_VG4_4Z2Z_H, AArch64::FMAXNM_VG4_4Z4Z_H,
6708 AArch64::FMAXNM_VG4_4Z4Z_S, AArch64::FMAXNM_VG4_4Z4Z_D}))
6709 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6710 return;
6711 case Intrinsic::aarch64_sve_fminnm_x2:
6713 Node->getValueType(0),
6714 {AArch64::BFMINNM_VG2_2Z2Z_H, AArch64::FMINNM_VG2_2Z2Z_H,
6715 AArch64::FMINNM_VG2_2Z2Z_S, AArch64::FMINNM_VG2_2Z2Z_D}))
6716 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6717 return;
6718 case Intrinsic::aarch64_sve_fminnm_x4:
6720 Node->getValueType(0),
6721 {AArch64::BFMINNM_VG4_4Z2Z_H, AArch64::FMINNM_VG4_4Z4Z_H,
6722 AArch64::FMINNM_VG4_4Z4Z_S, AArch64::FMINNM_VG4_4Z4Z_D}))
6723 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6724 return;
6725 case Intrinsic::aarch64_sve_aese_lane_x2:
6726 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESE_2ZZI_B);
6727 return;
6728 case Intrinsic::aarch64_sve_aesd_lane_x2:
6729 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESD_2ZZI_B);
6730 return;
6731 case Intrinsic::aarch64_sve_aesemc_lane_x2:
6732 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESEMC_2ZZI_B);
6733 return;
6734 case Intrinsic::aarch64_sve_aesdimc_lane_x2:
6735 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::AESDIMC_2ZZI_B);
6736 return;
6737 case Intrinsic::aarch64_sve_aese_lane_x4:
6738 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESE_4ZZI_B);
6739 return;
6740 case Intrinsic::aarch64_sve_aesd_lane_x4:
6741 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESD_4ZZI_B);
6742 return;
6743 case Intrinsic::aarch64_sve_aesemc_lane_x4:
6744 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESEMC_4ZZI_B);
6745 return;
6746 case Intrinsic::aarch64_sve_aesdimc_lane_x4:
6747 SelectDestructiveMultiIntrinsic(Node, 4, false, AArch64::AESDIMC_4ZZI_B);
6748 return;
6749 case Intrinsic::aarch64_sve_pmlal_pair_x2:
6750 SelectDestructiveMultiIntrinsic(Node, 2, false, AArch64::PMLAL_2ZZZ_Q);
6751 return;
6752 case Intrinsic::aarch64_sve_pmull_pair_x2: {
6753 SDLoc DL(Node);
6754 SmallVector<SDValue, 4> Regs(Node->ops().slice(1, 2));
6755 SDNode *Res =
6756 CurDAG->getMachineNode(AArch64::PMULL_2ZZZ_Q, DL, MVT::Untyped, Regs);
6757 SDValue SuperReg = SDValue(Res, 0);
6758 for (unsigned I = 0; I < 2; I++)
6759 ReplaceUses(SDValue(Node, I),
6760 CurDAG->getTargetExtractSubreg(AArch64::zsub0 + I, DL, VT,
6761 SuperReg));
6762 CurDAG->RemoveDeadNode(Node);
6763 return;
6764 }
6765 case Intrinsic::aarch64_sve_fscale_x4:
6766 SelectDestructiveMultiIntrinsic(Node, 4, true, AArch64::BFSCALE_4Z4Z);
6767 return;
6768 case Intrinsic::aarch64_sve_fscale_x2:
6769 SelectDestructiveMultiIntrinsic(Node, 2, true, AArch64::BFSCALE_2Z2Z);
6770 return;
6771 case Intrinsic::aarch64_sve_fmul_x4:
6773 Node->getValueType(0),
6774 {AArch64::BFMUL_4Z4Z, AArch64::FMUL_4Z4Z_H, AArch64::FMUL_4Z4Z_S,
6775 AArch64::FMUL_4Z4Z_D}))
6776 SelectDestructiveMultiIntrinsic(Node, 4, true, Op);
6777 return;
6778 case Intrinsic::aarch64_sve_fmul_x2:
6780 Node->getValueType(0),
6781 {AArch64::BFMUL_2Z2Z, AArch64::FMUL_2Z2Z_H, AArch64::FMUL_2Z2Z_S,
6782 AArch64::FMUL_2Z2Z_D}))
6783 SelectDestructiveMultiIntrinsic(Node, 2, true, Op);
6784 return;
6785 case Intrinsic::aarch64_sve_fcvtzs_x2:
6786 SelectCVTIntrinsic(Node, 2, AArch64::FCVTZS_2Z2Z_StoS);
6787 return;
6788 case Intrinsic::aarch64_sve_scvtf_x2:
6789 SelectCVTIntrinsic(Node, 2, AArch64::SCVTF_2Z2Z_StoS);
6790 return;
6791 case Intrinsic::aarch64_sve_fcvtzu_x2:
6792 SelectCVTIntrinsic(Node, 2, AArch64::FCVTZU_2Z2Z_StoS);
6793 return;
6794 case Intrinsic::aarch64_sve_ucvtf_x2:
6795 SelectCVTIntrinsic(Node, 2, AArch64::UCVTF_2Z2Z_StoS);
6796 return;
6797 case Intrinsic::aarch64_sve_fcvtzs_x4:
6798 SelectCVTIntrinsic(Node, 4, AArch64::FCVTZS_4Z4Z_StoS);
6799 return;
6800 case Intrinsic::aarch64_sve_scvtf_x4:
6801 SelectCVTIntrinsic(Node, 4, AArch64::SCVTF_4Z4Z_StoS);
6802 return;
6803 case Intrinsic::aarch64_sve_fcvtzu_x4:
6804 SelectCVTIntrinsic(Node, 4, AArch64::FCVTZU_4Z4Z_StoS);
6805 return;
6806 case Intrinsic::aarch64_sve_ucvtf_x4:
6807 SelectCVTIntrinsic(Node, 4, AArch64::UCVTF_4Z4Z_StoS);
6808 return;
6809 case Intrinsic::aarch64_sve_fcvt_widen_x2:
6810 SelectUnaryMultiIntrinsic(Node, 2, false, AArch64::FCVT_2ZZ_H_S);
6811 return;
6812 case Intrinsic::aarch64_sve_fcvtl_widen_x2:
6813 SelectUnaryMultiIntrinsic(Node, 2, false, AArch64::FCVTL_2ZZ_H_S);
6814 return;
6815 case Intrinsic::aarch64_sve_sclamp_single_x2:
6817 Node->getValueType(0),
6818 {AArch64::SCLAMP_VG2_2Z2Z_B, AArch64::SCLAMP_VG2_2Z2Z_H,
6819 AArch64::SCLAMP_VG2_2Z2Z_S, AArch64::SCLAMP_VG2_2Z2Z_D}))
6820 SelectClamp(Node, 2, Op);
6821 return;
6822 case Intrinsic::aarch64_sve_uclamp_single_x2:
6824 Node->getValueType(0),
6825 {AArch64::UCLAMP_VG2_2Z2Z_B, AArch64::UCLAMP_VG2_2Z2Z_H,
6826 AArch64::UCLAMP_VG2_2Z2Z_S, AArch64::UCLAMP_VG2_2Z2Z_D}))
6827 SelectClamp(Node, 2, Op);
6828 return;
6829 case Intrinsic::aarch64_sve_fclamp_single_x2:
6831 Node->getValueType(0),
6832 {0, AArch64::FCLAMP_VG2_2Z2Z_H, AArch64::FCLAMP_VG2_2Z2Z_S,
6833 AArch64::FCLAMP_VG2_2Z2Z_D}))
6834 SelectClamp(Node, 2, Op);
6835 return;
6836 case Intrinsic::aarch64_sve_bfclamp_single_x2:
6837 SelectClamp(Node, 2, AArch64::BFCLAMP_VG2_2ZZZ_H);
6838 return;
6839 case Intrinsic::aarch64_sve_sclamp_single_x4:
6841 Node->getValueType(0),
6842 {AArch64::SCLAMP_VG4_4Z4Z_B, AArch64::SCLAMP_VG4_4Z4Z_H,
6843 AArch64::SCLAMP_VG4_4Z4Z_S, AArch64::SCLAMP_VG4_4Z4Z_D}))
6844 SelectClamp(Node, 4, Op);
6845 return;
6846 case Intrinsic::aarch64_sve_uclamp_single_x4:
6848 Node->getValueType(0),
6849 {AArch64::UCLAMP_VG4_4Z4Z_B, AArch64::UCLAMP_VG4_4Z4Z_H,
6850 AArch64::UCLAMP_VG4_4Z4Z_S, AArch64::UCLAMP_VG4_4Z4Z_D}))
6851 SelectClamp(Node, 4, Op);
6852 return;
6853 case Intrinsic::aarch64_sve_fclamp_single_x4:
6855 Node->getValueType(0),
6856 {0, AArch64::FCLAMP_VG4_4Z4Z_H, AArch64::FCLAMP_VG4_4Z4Z_S,
6857 AArch64::FCLAMP_VG4_4Z4Z_D}))
6858 SelectClamp(Node, 4, Op);
6859 return;
6860 case Intrinsic::aarch64_sve_bfclamp_single_x4:
6861 SelectClamp(Node, 4, AArch64::BFCLAMP_VG4_4ZZZ_H);
6862 return;
6863 case Intrinsic::aarch64_sve_add_single_x2:
6865 Node->getValueType(0),
6866 {AArch64::ADD_VG2_2ZZ_B, AArch64::ADD_VG2_2ZZ_H,
6867 AArch64::ADD_VG2_2ZZ_S, AArch64::ADD_VG2_2ZZ_D}))
6868 SelectDestructiveMultiIntrinsic(Node, 2, false, Op);
6869 return;
6870 case Intrinsic::aarch64_sve_add_single_x4:
6872 Node->getValueType(0),
6873 {AArch64::ADD_VG4_4ZZ_B, AArch64::ADD_VG4_4ZZ_H,
6874 AArch64::ADD_VG4_4ZZ_S, AArch64::ADD_VG4_4ZZ_D}))
6875 SelectDestructiveMultiIntrinsic(Node, 4, false, Op);
6876 return;
6877 case Intrinsic::aarch64_sve_zip_x2:
6879 Node->getValueType(0),
6880 {AArch64::ZIP_VG2_2ZZZ_B, AArch64::ZIP_VG2_2ZZZ_H,
6881 AArch64::ZIP_VG2_2ZZZ_S, AArch64::ZIP_VG2_2ZZZ_D}))
6882 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6883 return;
6884 case Intrinsic::aarch64_sve_zipq_x2:
6885 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false,
6886 AArch64::ZIP_VG2_2ZZZ_Q);
6887 return;
6888 case Intrinsic::aarch64_sve_zip_x4:
6890 Node->getValueType(0),
6891 {AArch64::ZIP_VG4_4Z4Z_B, AArch64::ZIP_VG4_4Z4Z_H,
6892 AArch64::ZIP_VG4_4Z4Z_S, AArch64::ZIP_VG4_4Z4Z_D}))
6893 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6894 return;
6895 case Intrinsic::aarch64_sve_zipq_x4:
6896 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true,
6897 AArch64::ZIP_VG4_4Z4Z_Q);
6898 return;
6899 case Intrinsic::aarch64_sve_uzp_x2:
6901 Node->getValueType(0),
6902 {AArch64::UZP_VG2_2ZZZ_B, AArch64::UZP_VG2_2ZZZ_H,
6903 AArch64::UZP_VG2_2ZZZ_S, AArch64::UZP_VG2_2ZZZ_D}))
6904 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6905 return;
6906 case Intrinsic::aarch64_sve_uzpq_x2:
6907 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false,
6908 AArch64::UZP_VG2_2ZZZ_Q);
6909 return;
6910 case Intrinsic::aarch64_sve_uzp_x4:
6912 Node->getValueType(0),
6913 {AArch64::UZP_VG4_4Z4Z_B, AArch64::UZP_VG4_4Z4Z_H,
6914 AArch64::UZP_VG4_4Z4Z_S, AArch64::UZP_VG4_4Z4Z_D}))
6915 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6916 return;
6917 case Intrinsic::aarch64_sve_uzpq_x4:
6918 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true,
6919 AArch64::UZP_VG4_4Z4Z_Q);
6920 return;
6921 case Intrinsic::aarch64_sve_sel_x2:
6923 Node->getValueType(0),
6924 {AArch64::SEL_VG2_2ZC2Z2Z_B, AArch64::SEL_VG2_2ZC2Z2Z_H,
6925 AArch64::SEL_VG2_2ZC2Z2Z_S, AArch64::SEL_VG2_2ZC2Z2Z_D}))
6926 SelectDestructiveMultiIntrinsic(Node, 2, true, Op, /*HasPred=*/true);
6927 return;
6928 case Intrinsic::aarch64_sve_sel_x4:
6930 Node->getValueType(0),
6931 {AArch64::SEL_VG4_4ZC4Z4Z_B, AArch64::SEL_VG4_4ZC4Z4Z_H,
6932 AArch64::SEL_VG4_4ZC4Z4Z_S, AArch64::SEL_VG4_4ZC4Z4Z_D}))
6933 SelectDestructiveMultiIntrinsic(Node, 4, true, Op, /*HasPred=*/true);
6934 return;
6935 case Intrinsic::aarch64_sve_frinta_x2:
6936 SelectFrintFromVT(Node, 2, AArch64::FRINTA_2Z2Z_S);
6937 return;
6938 case Intrinsic::aarch64_sve_frinta_x4:
6939 SelectFrintFromVT(Node, 4, AArch64::FRINTA_4Z4Z_S);
6940 return;
6941 case Intrinsic::aarch64_sve_frintm_x2:
6942 SelectFrintFromVT(Node, 2, AArch64::FRINTM_2Z2Z_S);
6943 return;
6944 case Intrinsic::aarch64_sve_frintm_x4:
6945 SelectFrintFromVT(Node, 4, AArch64::FRINTM_4Z4Z_S);
6946 return;
6947 case Intrinsic::aarch64_sve_frintn_x2:
6948 SelectFrintFromVT(Node, 2, AArch64::FRINTN_2Z2Z_S);
6949 return;
6950 case Intrinsic::aarch64_sve_frintn_x4:
6951 SelectFrintFromVT(Node, 4, AArch64::FRINTN_4Z4Z_S);
6952 return;
6953 case Intrinsic::aarch64_sve_frintp_x2:
6954 SelectFrintFromVT(Node, 2, AArch64::FRINTP_2Z2Z_S);
6955 return;
6956 case Intrinsic::aarch64_sve_frintp_x4:
6957 SelectFrintFromVT(Node, 4, AArch64::FRINTP_4Z4Z_S);
6958 return;
6959 case Intrinsic::aarch64_sve_sunpk_x2:
6961 Node->getValueType(0),
6962 {0, AArch64::SUNPK_VG2_2ZZ_H, AArch64::SUNPK_VG2_2ZZ_S,
6963 AArch64::SUNPK_VG2_2ZZ_D}))
6964 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6965 return;
6966 case Intrinsic::aarch64_sve_uunpk_x2:
6968 Node->getValueType(0),
6969 {0, AArch64::UUNPK_VG2_2ZZ_H, AArch64::UUNPK_VG2_2ZZ_S,
6970 AArch64::UUNPK_VG2_2ZZ_D}))
6971 SelectUnaryMultiIntrinsic(Node, 2, /*IsTupleInput=*/false, Op);
6972 return;
6973 case Intrinsic::aarch64_sve_sunpk_x4:
6975 Node->getValueType(0),
6976 {0, AArch64::SUNPK_VG4_4Z2Z_H, AArch64::SUNPK_VG4_4Z2Z_S,
6977 AArch64::SUNPK_VG4_4Z2Z_D}))
6978 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6979 return;
6980 case Intrinsic::aarch64_sve_uunpk_x4:
6982 Node->getValueType(0),
6983 {0, AArch64::UUNPK_VG4_4Z2Z_H, AArch64::UUNPK_VG4_4Z2Z_S,
6984 AArch64::UUNPK_VG4_4Z2Z_D}))
6985 SelectUnaryMultiIntrinsic(Node, 4, /*IsTupleInput=*/true, Op);
6986 return;
6987 case Intrinsic::aarch64_sve_pext_x2: {
6989 Node->getValueType(0),
6990 {AArch64::PEXT_2PCI_B, AArch64::PEXT_2PCI_H, AArch64::PEXT_2PCI_S,
6991 AArch64::PEXT_2PCI_D}))
6992 SelectPExtPair(Node, Op);
6993 return;
6994 }
6995 }
6996 break;
6997 }
6998 case ISD::INTRINSIC_VOID: {
6999 unsigned IntNo = Node->getConstantOperandVal(1);
7000 if (Node->getNumOperands() >= 3)
7001 VT = Node->getOperand(2)->getValueType(0);
7002 switch (IntNo) {
7003 default:
7004 break;
7005 case Intrinsic::aarch64_neon_st1x2: {
7006 if (VT == MVT::v8i8) {
7007 SelectStore(Node, 2, AArch64::ST1Twov8b);
7008 return;
7009 } else if (VT == MVT::v16i8) {
7010 SelectStore(Node, 2, AArch64::ST1Twov16b);
7011 return;
7012 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7013 VT == MVT::v4bf16) {
7014 SelectStore(Node, 2, AArch64::ST1Twov4h);
7015 return;
7016 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7017 VT == MVT::v8bf16) {
7018 SelectStore(Node, 2, AArch64::ST1Twov8h);
7019 return;
7020 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7021 SelectStore(Node, 2, AArch64::ST1Twov2s);
7022 return;
7023 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7024 SelectStore(Node, 2, AArch64::ST1Twov4s);
7025 return;
7026 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7027 SelectStore(Node, 2, AArch64::ST1Twov2d);
7028 return;
7029 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7030 SelectStore(Node, 2, AArch64::ST1Twov1d);
7031 return;
7032 }
7033 break;
7034 }
7035 case Intrinsic::aarch64_neon_st1x3: {
7036 if (VT == MVT::v8i8) {
7037 SelectStore(Node, 3, AArch64::ST1Threev8b);
7038 return;
7039 } else if (VT == MVT::v16i8) {
7040 SelectStore(Node, 3, AArch64::ST1Threev16b);
7041 return;
7042 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7043 VT == MVT::v4bf16) {
7044 SelectStore(Node, 3, AArch64::ST1Threev4h);
7045 return;
7046 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7047 VT == MVT::v8bf16) {
7048 SelectStore(Node, 3, AArch64::ST1Threev8h);
7049 return;
7050 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7051 SelectStore(Node, 3, AArch64::ST1Threev2s);
7052 return;
7053 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7054 SelectStore(Node, 3, AArch64::ST1Threev4s);
7055 return;
7056 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7057 SelectStore(Node, 3, AArch64::ST1Threev2d);
7058 return;
7059 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7060 SelectStore(Node, 3, AArch64::ST1Threev1d);
7061 return;
7062 }
7063 break;
7064 }
7065 case Intrinsic::aarch64_neon_st1x4: {
7066 if (VT == MVT::v8i8) {
7067 SelectStore(Node, 4, AArch64::ST1Fourv8b);
7068 return;
7069 } else if (VT == MVT::v16i8) {
7070 SelectStore(Node, 4, AArch64::ST1Fourv16b);
7071 return;
7072 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7073 VT == MVT::v4bf16) {
7074 SelectStore(Node, 4, AArch64::ST1Fourv4h);
7075 return;
7076 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7077 VT == MVT::v8bf16) {
7078 SelectStore(Node, 4, AArch64::ST1Fourv8h);
7079 return;
7080 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7081 SelectStore(Node, 4, AArch64::ST1Fourv2s);
7082 return;
7083 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7084 SelectStore(Node, 4, AArch64::ST1Fourv4s);
7085 return;
7086 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7087 SelectStore(Node, 4, AArch64::ST1Fourv2d);
7088 return;
7089 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7090 SelectStore(Node, 4, AArch64::ST1Fourv1d);
7091 return;
7092 }
7093 break;
7094 }
7095 case Intrinsic::aarch64_neon_st2: {
7096 if (VT == MVT::v8i8) {
7097 SelectStore(Node, 2, AArch64::ST2Twov8b);
7098 return;
7099 } else if (VT == MVT::v16i8) {
7100 SelectStore(Node, 2, AArch64::ST2Twov16b);
7101 return;
7102 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7103 VT == MVT::v4bf16) {
7104 SelectStore(Node, 2, AArch64::ST2Twov4h);
7105 return;
7106 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7107 VT == MVT::v8bf16) {
7108 SelectStore(Node, 2, AArch64::ST2Twov8h);
7109 return;
7110 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7111 SelectStore(Node, 2, AArch64::ST2Twov2s);
7112 return;
7113 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7114 SelectStore(Node, 2, AArch64::ST2Twov4s);
7115 return;
7116 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7117 SelectStore(Node, 2, AArch64::ST2Twov2d);
7118 return;
7119 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7120 SelectStore(Node, 2, AArch64::ST1Twov1d);
7121 return;
7122 }
7123 break;
7124 }
7125 case Intrinsic::aarch64_neon_st3: {
7126 if (VT == MVT::v8i8) {
7127 SelectStore(Node, 3, AArch64::ST3Threev8b);
7128 return;
7129 } else if (VT == MVT::v16i8) {
7130 SelectStore(Node, 3, AArch64::ST3Threev16b);
7131 return;
7132 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7133 VT == MVT::v4bf16) {
7134 SelectStore(Node, 3, AArch64::ST3Threev4h);
7135 return;
7136 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7137 VT == MVT::v8bf16) {
7138 SelectStore(Node, 3, AArch64::ST3Threev8h);
7139 return;
7140 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7141 SelectStore(Node, 3, AArch64::ST3Threev2s);
7142 return;
7143 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7144 SelectStore(Node, 3, AArch64::ST3Threev4s);
7145 return;
7146 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7147 SelectStore(Node, 3, AArch64::ST3Threev2d);
7148 return;
7149 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7150 SelectStore(Node, 3, AArch64::ST1Threev1d);
7151 return;
7152 }
7153 break;
7154 }
7155 case Intrinsic::aarch64_neon_st4: {
7156 if (VT == MVT::v8i8) {
7157 SelectStore(Node, 4, AArch64::ST4Fourv8b);
7158 return;
7159 } else if (VT == MVT::v16i8) {
7160 SelectStore(Node, 4, AArch64::ST4Fourv16b);
7161 return;
7162 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 ||
7163 VT == MVT::v4bf16) {
7164 SelectStore(Node, 4, AArch64::ST4Fourv4h);
7165 return;
7166 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 ||
7167 VT == MVT::v8bf16) {
7168 SelectStore(Node, 4, AArch64::ST4Fourv8h);
7169 return;
7170 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7171 SelectStore(Node, 4, AArch64::ST4Fourv2s);
7172 return;
7173 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7174 SelectStore(Node, 4, AArch64::ST4Fourv4s);
7175 return;
7176 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7177 SelectStore(Node, 4, AArch64::ST4Fourv2d);
7178 return;
7179 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7180 SelectStore(Node, 4, AArch64::ST1Fourv1d);
7181 return;
7182 }
7183 break;
7184 }
7185 case Intrinsic::aarch64_neon_st2lane: {
7186 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7187 SelectStoreLane(Node, 2, AArch64::ST2i8);
7188 return;
7189 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7190 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7191 SelectStoreLane(Node, 2, AArch64::ST2i16);
7192 return;
7193 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7194 VT == MVT::v2f32) {
7195 SelectStoreLane(Node, 2, AArch64::ST2i32);
7196 return;
7197 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7198 VT == MVT::v1f64) {
7199 SelectStoreLane(Node, 2, AArch64::ST2i64);
7200 return;
7201 }
7202 break;
7203 }
7204 case Intrinsic::aarch64_neon_st3lane: {
7205 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7206 SelectStoreLane(Node, 3, AArch64::ST3i8);
7207 return;
7208 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7209 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7210 SelectStoreLane(Node, 3, AArch64::ST3i16);
7211 return;
7212 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7213 VT == MVT::v2f32) {
7214 SelectStoreLane(Node, 3, AArch64::ST3i32);
7215 return;
7216 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7217 VT == MVT::v1f64) {
7218 SelectStoreLane(Node, 3, AArch64::ST3i64);
7219 return;
7220 }
7221 break;
7222 }
7223 case Intrinsic::aarch64_neon_st4lane: {
7224 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7225 SelectStoreLane(Node, 4, AArch64::ST4i8);
7226 return;
7227 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7228 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7229 SelectStoreLane(Node, 4, AArch64::ST4i16);
7230 return;
7231 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7232 VT == MVT::v2f32) {
7233 SelectStoreLane(Node, 4, AArch64::ST4i32);
7234 return;
7235 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7236 VT == MVT::v1f64) {
7237 SelectStoreLane(Node, 4, AArch64::ST4i64);
7238 return;
7239 }
7240 break;
7241 }
7242 case Intrinsic::aarch64_sve_st2q: {
7243 SelectPredicatedStore(Node, 2, 4, AArch64::ST2Q, AArch64::ST2Q_IMM);
7244 return;
7245 }
7246 case Intrinsic::aarch64_sve_st3q: {
7247 SelectPredicatedStore(Node, 3, 4, AArch64::ST3Q, AArch64::ST3Q_IMM);
7248 return;
7249 }
7250 case Intrinsic::aarch64_sve_st4q: {
7251 SelectPredicatedStore(Node, 4, 4, AArch64::ST4Q, AArch64::ST4Q_IMM);
7252 return;
7253 }
7254 case Intrinsic::aarch64_sve_st2: {
7255 if (VT == MVT::nxv16i8) {
7256 SelectPredicatedStore(Node, 2, 0, AArch64::ST2B, AArch64::ST2B_IMM);
7257 return;
7258 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7259 VT == MVT::nxv8bf16) {
7260 SelectPredicatedStore(Node, 2, 1, AArch64::ST2H, AArch64::ST2H_IMM);
7261 return;
7262 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7263 SelectPredicatedStore(Node, 2, 2, AArch64::ST2W, AArch64::ST2W_IMM);
7264 return;
7265 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7266 SelectPredicatedStore(Node, 2, 3, AArch64::ST2D, AArch64::ST2D_IMM);
7267 return;
7268 }
7269 break;
7270 }
7271 case Intrinsic::aarch64_sve_st3: {
7272 if (VT == MVT::nxv16i8) {
7273 SelectPredicatedStore(Node, 3, 0, AArch64::ST3B, AArch64::ST3B_IMM);
7274 return;
7275 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7276 VT == MVT::nxv8bf16) {
7277 SelectPredicatedStore(Node, 3, 1, AArch64::ST3H, AArch64::ST3H_IMM);
7278 return;
7279 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7280 SelectPredicatedStore(Node, 3, 2, AArch64::ST3W, AArch64::ST3W_IMM);
7281 return;
7282 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7283 SelectPredicatedStore(Node, 3, 3, AArch64::ST3D, AArch64::ST3D_IMM);
7284 return;
7285 }
7286 break;
7287 }
7288 case Intrinsic::aarch64_sve_st4: {
7289 if (VT == MVT::nxv16i8) {
7290 SelectPredicatedStore(Node, 4, 0, AArch64::ST4B, AArch64::ST4B_IMM);
7291 return;
7292 } else if (VT == MVT::nxv8i16 || VT == MVT::nxv8f16 ||
7293 VT == MVT::nxv8bf16) {
7294 SelectPredicatedStore(Node, 4, 1, AArch64::ST4H, AArch64::ST4H_IMM);
7295 return;
7296 } else if (VT == MVT::nxv4i32 || VT == MVT::nxv4f32) {
7297 SelectPredicatedStore(Node, 4, 2, AArch64::ST4W, AArch64::ST4W_IMM);
7298 return;
7299 } else if (VT == MVT::nxv2i64 || VT == MVT::nxv2f64) {
7300 SelectPredicatedStore(Node, 4, 3, AArch64::ST4D, AArch64::ST4D_IMM);
7301 return;
7302 }
7303 break;
7304 }
7305 }
7306 break;
7307 }
7308 case AArch64ISD::LD2post: {
7309 if (VT == MVT::v8i8) {
7310 SelectPostLoad(Node, 2, AArch64::LD2Twov8b_POST, AArch64::dsub0);
7311 return;
7312 } else if (VT == MVT::v16i8) {
7313 SelectPostLoad(Node, 2, AArch64::LD2Twov16b_POST, AArch64::qsub0);
7314 return;
7315 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7316 SelectPostLoad(Node, 2, AArch64::LD2Twov4h_POST, AArch64::dsub0);
7317 return;
7318 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7319 SelectPostLoad(Node, 2, AArch64::LD2Twov8h_POST, AArch64::qsub0);
7320 return;
7321 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7322 SelectPostLoad(Node, 2, AArch64::LD2Twov2s_POST, AArch64::dsub0);
7323 return;
7324 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7325 SelectPostLoad(Node, 2, AArch64::LD2Twov4s_POST, AArch64::qsub0);
7326 return;
7327 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7328 SelectPostLoad(Node, 2, AArch64::LD1Twov1d_POST, AArch64::dsub0);
7329 return;
7330 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7331 SelectPostLoad(Node, 2, AArch64::LD2Twov2d_POST, AArch64::qsub0);
7332 return;
7333 }
7334 break;
7335 }
7336 case AArch64ISD::LD3post: {
7337 if (VT == MVT::v8i8) {
7338 SelectPostLoad(Node, 3, AArch64::LD3Threev8b_POST, AArch64::dsub0);
7339 return;
7340 } else if (VT == MVT::v16i8) {
7341 SelectPostLoad(Node, 3, AArch64::LD3Threev16b_POST, AArch64::qsub0);
7342 return;
7343 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7344 SelectPostLoad(Node, 3, AArch64::LD3Threev4h_POST, AArch64::dsub0);
7345 return;
7346 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7347 SelectPostLoad(Node, 3, AArch64::LD3Threev8h_POST, AArch64::qsub0);
7348 return;
7349 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7350 SelectPostLoad(Node, 3, AArch64::LD3Threev2s_POST, AArch64::dsub0);
7351 return;
7352 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7353 SelectPostLoad(Node, 3, AArch64::LD3Threev4s_POST, AArch64::qsub0);
7354 return;
7355 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7356 SelectPostLoad(Node, 3, AArch64::LD1Threev1d_POST, AArch64::dsub0);
7357 return;
7358 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7359 SelectPostLoad(Node, 3, AArch64::LD3Threev2d_POST, AArch64::qsub0);
7360 return;
7361 }
7362 break;
7363 }
7364 case AArch64ISD::LD4post: {
7365 if (VT == MVT::v8i8) {
7366 SelectPostLoad(Node, 4, AArch64::LD4Fourv8b_POST, AArch64::dsub0);
7367 return;
7368 } else if (VT == MVT::v16i8) {
7369 SelectPostLoad(Node, 4, AArch64::LD4Fourv16b_POST, AArch64::qsub0);
7370 return;
7371 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7372 SelectPostLoad(Node, 4, AArch64::LD4Fourv4h_POST, AArch64::dsub0);
7373 return;
7374 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7375 SelectPostLoad(Node, 4, AArch64::LD4Fourv8h_POST, AArch64::qsub0);
7376 return;
7377 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7378 SelectPostLoad(Node, 4, AArch64::LD4Fourv2s_POST, AArch64::dsub0);
7379 return;
7380 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7381 SelectPostLoad(Node, 4, AArch64::LD4Fourv4s_POST, AArch64::qsub0);
7382 return;
7383 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7384 SelectPostLoad(Node, 4, AArch64::LD1Fourv1d_POST, AArch64::dsub0);
7385 return;
7386 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7387 SelectPostLoad(Node, 4, AArch64::LD4Fourv2d_POST, AArch64::qsub0);
7388 return;
7389 }
7390 break;
7391 }
7392 case AArch64ISD::LD1x2post: {
7393 if (VT == MVT::v8i8) {
7394 SelectPostLoad(Node, 2, AArch64::LD1Twov8b_POST, AArch64::dsub0);
7395 return;
7396 } else if (VT == MVT::v16i8) {
7397 SelectPostLoad(Node, 2, AArch64::LD1Twov16b_POST, AArch64::qsub0);
7398 return;
7399 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7400 SelectPostLoad(Node, 2, AArch64::LD1Twov4h_POST, AArch64::dsub0);
7401 return;
7402 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7403 SelectPostLoad(Node, 2, AArch64::LD1Twov8h_POST, AArch64::qsub0);
7404 return;
7405 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7406 SelectPostLoad(Node, 2, AArch64::LD1Twov2s_POST, AArch64::dsub0);
7407 return;
7408 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7409 SelectPostLoad(Node, 2, AArch64::LD1Twov4s_POST, AArch64::qsub0);
7410 return;
7411 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7412 SelectPostLoad(Node, 2, AArch64::LD1Twov1d_POST, AArch64::dsub0);
7413 return;
7414 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7415 SelectPostLoad(Node, 2, AArch64::LD1Twov2d_POST, AArch64::qsub0);
7416 return;
7417 }
7418 break;
7419 }
7420 case AArch64ISD::LD1x3post: {
7421 if (VT == MVT::v8i8) {
7422 SelectPostLoad(Node, 3, AArch64::LD1Threev8b_POST, AArch64::dsub0);
7423 return;
7424 } else if (VT == MVT::v16i8) {
7425 SelectPostLoad(Node, 3, AArch64::LD1Threev16b_POST, AArch64::qsub0);
7426 return;
7427 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7428 SelectPostLoad(Node, 3, AArch64::LD1Threev4h_POST, AArch64::dsub0);
7429 return;
7430 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7431 SelectPostLoad(Node, 3, AArch64::LD1Threev8h_POST, AArch64::qsub0);
7432 return;
7433 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7434 SelectPostLoad(Node, 3, AArch64::LD1Threev2s_POST, AArch64::dsub0);
7435 return;
7436 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7437 SelectPostLoad(Node, 3, AArch64::LD1Threev4s_POST, AArch64::qsub0);
7438 return;
7439 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7440 SelectPostLoad(Node, 3, AArch64::LD1Threev1d_POST, AArch64::dsub0);
7441 return;
7442 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7443 SelectPostLoad(Node, 3, AArch64::LD1Threev2d_POST, AArch64::qsub0);
7444 return;
7445 }
7446 break;
7447 }
7448 case AArch64ISD::LD1x4post: {
7449 if (VT == MVT::v8i8) {
7450 SelectPostLoad(Node, 4, AArch64::LD1Fourv8b_POST, AArch64::dsub0);
7451 return;
7452 } else if (VT == MVT::v16i8) {
7453 SelectPostLoad(Node, 4, AArch64::LD1Fourv16b_POST, AArch64::qsub0);
7454 return;
7455 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7456 SelectPostLoad(Node, 4, AArch64::LD1Fourv4h_POST, AArch64::dsub0);
7457 return;
7458 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7459 SelectPostLoad(Node, 4, AArch64::LD1Fourv8h_POST, AArch64::qsub0);
7460 return;
7461 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7462 SelectPostLoad(Node, 4, AArch64::LD1Fourv2s_POST, AArch64::dsub0);
7463 return;
7464 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7465 SelectPostLoad(Node, 4, AArch64::LD1Fourv4s_POST, AArch64::qsub0);
7466 return;
7467 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7468 SelectPostLoad(Node, 4, AArch64::LD1Fourv1d_POST, AArch64::dsub0);
7469 return;
7470 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7471 SelectPostLoad(Node, 4, AArch64::LD1Fourv2d_POST, AArch64::qsub0);
7472 return;
7473 }
7474 break;
7475 }
7476 case AArch64ISD::LD1DUPpost: {
7477 if (VT == MVT::v8i8) {
7478 SelectPostLoad(Node, 1, AArch64::LD1Rv8b_POST, AArch64::dsub0);
7479 return;
7480 } else if (VT == MVT::v16i8) {
7481 SelectPostLoad(Node, 1, AArch64::LD1Rv16b_POST, AArch64::qsub0);
7482 return;
7483 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7484 SelectPostLoad(Node, 1, AArch64::LD1Rv4h_POST, AArch64::dsub0);
7485 return;
7486 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7487 SelectPostLoad(Node, 1, AArch64::LD1Rv8h_POST, AArch64::qsub0);
7488 return;
7489 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7490 SelectPostLoad(Node, 1, AArch64::LD1Rv2s_POST, AArch64::dsub0);
7491 return;
7492 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7493 SelectPostLoad(Node, 1, AArch64::LD1Rv4s_POST, AArch64::qsub0);
7494 return;
7495 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7496 SelectPostLoad(Node, 1, AArch64::LD1Rv1d_POST, AArch64::dsub0);
7497 return;
7498 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7499 SelectPostLoad(Node, 1, AArch64::LD1Rv2d_POST, AArch64::qsub0);
7500 return;
7501 }
7502 break;
7503 }
7504 case AArch64ISD::LD2DUPpost: {
7505 if (VT == MVT::v8i8) {
7506 SelectPostLoad(Node, 2, AArch64::LD2Rv8b_POST, AArch64::dsub0);
7507 return;
7508 } else if (VT == MVT::v16i8) {
7509 SelectPostLoad(Node, 2, AArch64::LD2Rv16b_POST, AArch64::qsub0);
7510 return;
7511 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7512 SelectPostLoad(Node, 2, AArch64::LD2Rv4h_POST, AArch64::dsub0);
7513 return;
7514 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7515 SelectPostLoad(Node, 2, AArch64::LD2Rv8h_POST, AArch64::qsub0);
7516 return;
7517 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7518 SelectPostLoad(Node, 2, AArch64::LD2Rv2s_POST, AArch64::dsub0);
7519 return;
7520 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7521 SelectPostLoad(Node, 2, AArch64::LD2Rv4s_POST, AArch64::qsub0);
7522 return;
7523 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7524 SelectPostLoad(Node, 2, AArch64::LD2Rv1d_POST, AArch64::dsub0);
7525 return;
7526 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7527 SelectPostLoad(Node, 2, AArch64::LD2Rv2d_POST, AArch64::qsub0);
7528 return;
7529 }
7530 break;
7531 }
7532 case AArch64ISD::LD3DUPpost: {
7533 if (VT == MVT::v8i8) {
7534 SelectPostLoad(Node, 3, AArch64::LD3Rv8b_POST, AArch64::dsub0);
7535 return;
7536 } else if (VT == MVT::v16i8) {
7537 SelectPostLoad(Node, 3, AArch64::LD3Rv16b_POST, AArch64::qsub0);
7538 return;
7539 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7540 SelectPostLoad(Node, 3, AArch64::LD3Rv4h_POST, AArch64::dsub0);
7541 return;
7542 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7543 SelectPostLoad(Node, 3, AArch64::LD3Rv8h_POST, AArch64::qsub0);
7544 return;
7545 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7546 SelectPostLoad(Node, 3, AArch64::LD3Rv2s_POST, AArch64::dsub0);
7547 return;
7548 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7549 SelectPostLoad(Node, 3, AArch64::LD3Rv4s_POST, AArch64::qsub0);
7550 return;
7551 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7552 SelectPostLoad(Node, 3, AArch64::LD3Rv1d_POST, AArch64::dsub0);
7553 return;
7554 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7555 SelectPostLoad(Node, 3, AArch64::LD3Rv2d_POST, AArch64::qsub0);
7556 return;
7557 }
7558 break;
7559 }
7560 case AArch64ISD::LD4DUPpost: {
7561 if (VT == MVT::v8i8) {
7562 SelectPostLoad(Node, 4, AArch64::LD4Rv8b_POST, AArch64::dsub0);
7563 return;
7564 } else if (VT == MVT::v16i8) {
7565 SelectPostLoad(Node, 4, AArch64::LD4Rv16b_POST, AArch64::qsub0);
7566 return;
7567 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7568 SelectPostLoad(Node, 4, AArch64::LD4Rv4h_POST, AArch64::dsub0);
7569 return;
7570 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7571 SelectPostLoad(Node, 4, AArch64::LD4Rv8h_POST, AArch64::qsub0);
7572 return;
7573 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7574 SelectPostLoad(Node, 4, AArch64::LD4Rv2s_POST, AArch64::dsub0);
7575 return;
7576 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7577 SelectPostLoad(Node, 4, AArch64::LD4Rv4s_POST, AArch64::qsub0);
7578 return;
7579 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7580 SelectPostLoad(Node, 4, AArch64::LD4Rv1d_POST, AArch64::dsub0);
7581 return;
7582 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7583 SelectPostLoad(Node, 4, AArch64::LD4Rv2d_POST, AArch64::qsub0);
7584 return;
7585 }
7586 break;
7587 }
7588 case AArch64ISD::LD1LANEpost: {
7589 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7590 SelectPostLoadLane(Node, 1, AArch64::LD1i8_POST);
7591 return;
7592 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7593 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7594 SelectPostLoadLane(Node, 1, AArch64::LD1i16_POST);
7595 return;
7596 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7597 VT == MVT::v2f32) {
7598 SelectPostLoadLane(Node, 1, AArch64::LD1i32_POST);
7599 return;
7600 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7601 VT == MVT::v1f64) {
7602 SelectPostLoadLane(Node, 1, AArch64::LD1i64_POST);
7603 return;
7604 }
7605 break;
7606 }
7607 case AArch64ISD::LD2LANEpost: {
7608 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7609 SelectPostLoadLane(Node, 2, AArch64::LD2i8_POST);
7610 return;
7611 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7612 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7613 SelectPostLoadLane(Node, 2, AArch64::LD2i16_POST);
7614 return;
7615 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7616 VT == MVT::v2f32) {
7617 SelectPostLoadLane(Node, 2, AArch64::LD2i32_POST);
7618 return;
7619 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7620 VT == MVT::v1f64) {
7621 SelectPostLoadLane(Node, 2, AArch64::LD2i64_POST);
7622 return;
7623 }
7624 break;
7625 }
7626 case AArch64ISD::LD3LANEpost: {
7627 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7628 SelectPostLoadLane(Node, 3, AArch64::LD3i8_POST);
7629 return;
7630 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7631 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7632 SelectPostLoadLane(Node, 3, AArch64::LD3i16_POST);
7633 return;
7634 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7635 VT == MVT::v2f32) {
7636 SelectPostLoadLane(Node, 3, AArch64::LD3i32_POST);
7637 return;
7638 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7639 VT == MVT::v1f64) {
7640 SelectPostLoadLane(Node, 3, AArch64::LD3i64_POST);
7641 return;
7642 }
7643 break;
7644 }
7645 case AArch64ISD::LD4LANEpost: {
7646 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7647 SelectPostLoadLane(Node, 4, AArch64::LD4i8_POST);
7648 return;
7649 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7650 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7651 SelectPostLoadLane(Node, 4, AArch64::LD4i16_POST);
7652 return;
7653 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7654 VT == MVT::v2f32) {
7655 SelectPostLoadLane(Node, 4, AArch64::LD4i32_POST);
7656 return;
7657 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7658 VT == MVT::v1f64) {
7659 SelectPostLoadLane(Node, 4, AArch64::LD4i64_POST);
7660 return;
7661 }
7662 break;
7663 }
7664 case AArch64ISD::ST2post: {
7665 VT = Node->getOperand(1).getValueType();
7666 if (VT == MVT::v8i8) {
7667 SelectPostStore(Node, 2, AArch64::ST2Twov8b_POST);
7668 return;
7669 } else if (VT == MVT::v16i8) {
7670 SelectPostStore(Node, 2, AArch64::ST2Twov16b_POST);
7671 return;
7672 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7673 SelectPostStore(Node, 2, AArch64::ST2Twov4h_POST);
7674 return;
7675 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7676 SelectPostStore(Node, 2, AArch64::ST2Twov8h_POST);
7677 return;
7678 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7679 SelectPostStore(Node, 2, AArch64::ST2Twov2s_POST);
7680 return;
7681 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7682 SelectPostStore(Node, 2, AArch64::ST2Twov4s_POST);
7683 return;
7684 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7685 SelectPostStore(Node, 2, AArch64::ST2Twov2d_POST);
7686 return;
7687 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7688 SelectPostStore(Node, 2, AArch64::ST1Twov1d_POST);
7689 return;
7690 }
7691 break;
7692 }
7693 case AArch64ISD::ST3post: {
7694 VT = Node->getOperand(1).getValueType();
7695 if (VT == MVT::v8i8) {
7696 SelectPostStore(Node, 3, AArch64::ST3Threev8b_POST);
7697 return;
7698 } else if (VT == MVT::v16i8) {
7699 SelectPostStore(Node, 3, AArch64::ST3Threev16b_POST);
7700 return;
7701 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7702 SelectPostStore(Node, 3, AArch64::ST3Threev4h_POST);
7703 return;
7704 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7705 SelectPostStore(Node, 3, AArch64::ST3Threev8h_POST);
7706 return;
7707 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7708 SelectPostStore(Node, 3, AArch64::ST3Threev2s_POST);
7709 return;
7710 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7711 SelectPostStore(Node, 3, AArch64::ST3Threev4s_POST);
7712 return;
7713 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7714 SelectPostStore(Node, 3, AArch64::ST3Threev2d_POST);
7715 return;
7716 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7717 SelectPostStore(Node, 3, AArch64::ST1Threev1d_POST);
7718 return;
7719 }
7720 break;
7721 }
7722 case AArch64ISD::ST4post: {
7723 VT = Node->getOperand(1).getValueType();
7724 if (VT == MVT::v8i8) {
7725 SelectPostStore(Node, 4, AArch64::ST4Fourv8b_POST);
7726 return;
7727 } else if (VT == MVT::v16i8) {
7728 SelectPostStore(Node, 4, AArch64::ST4Fourv16b_POST);
7729 return;
7730 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7731 SelectPostStore(Node, 4, AArch64::ST4Fourv4h_POST);
7732 return;
7733 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7734 SelectPostStore(Node, 4, AArch64::ST4Fourv8h_POST);
7735 return;
7736 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7737 SelectPostStore(Node, 4, AArch64::ST4Fourv2s_POST);
7738 return;
7739 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7740 SelectPostStore(Node, 4, AArch64::ST4Fourv4s_POST);
7741 return;
7742 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7743 SelectPostStore(Node, 4, AArch64::ST4Fourv2d_POST);
7744 return;
7745 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7746 SelectPostStore(Node, 4, AArch64::ST1Fourv1d_POST);
7747 return;
7748 }
7749 break;
7750 }
7751 case AArch64ISD::ST1x2post: {
7752 VT = Node->getOperand(1).getValueType();
7753 if (VT == MVT::v8i8) {
7754 SelectPostStore(Node, 2, AArch64::ST1Twov8b_POST);
7755 return;
7756 } else if (VT == MVT::v16i8) {
7757 SelectPostStore(Node, 2, AArch64::ST1Twov16b_POST);
7758 return;
7759 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7760 SelectPostStore(Node, 2, AArch64::ST1Twov4h_POST);
7761 return;
7762 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7763 SelectPostStore(Node, 2, AArch64::ST1Twov8h_POST);
7764 return;
7765 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7766 SelectPostStore(Node, 2, AArch64::ST1Twov2s_POST);
7767 return;
7768 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7769 SelectPostStore(Node, 2, AArch64::ST1Twov4s_POST);
7770 return;
7771 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7772 SelectPostStore(Node, 2, AArch64::ST1Twov1d_POST);
7773 return;
7774 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7775 SelectPostStore(Node, 2, AArch64::ST1Twov2d_POST);
7776 return;
7777 }
7778 break;
7779 }
7780 case AArch64ISD::ST1x3post: {
7781 VT = Node->getOperand(1).getValueType();
7782 if (VT == MVT::v8i8) {
7783 SelectPostStore(Node, 3, AArch64::ST1Threev8b_POST);
7784 return;
7785 } else if (VT == MVT::v16i8) {
7786 SelectPostStore(Node, 3, AArch64::ST1Threev16b_POST);
7787 return;
7788 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7789 SelectPostStore(Node, 3, AArch64::ST1Threev4h_POST);
7790 return;
7791 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16 ) {
7792 SelectPostStore(Node, 3, AArch64::ST1Threev8h_POST);
7793 return;
7794 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7795 SelectPostStore(Node, 3, AArch64::ST1Threev2s_POST);
7796 return;
7797 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7798 SelectPostStore(Node, 3, AArch64::ST1Threev4s_POST);
7799 return;
7800 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7801 SelectPostStore(Node, 3, AArch64::ST1Threev1d_POST);
7802 return;
7803 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7804 SelectPostStore(Node, 3, AArch64::ST1Threev2d_POST);
7805 return;
7806 }
7807 break;
7808 }
7809 case AArch64ISD::ST1x4post: {
7810 VT = Node->getOperand(1).getValueType();
7811 if (VT == MVT::v8i8) {
7812 SelectPostStore(Node, 4, AArch64::ST1Fourv8b_POST);
7813 return;
7814 } else if (VT == MVT::v16i8) {
7815 SelectPostStore(Node, 4, AArch64::ST1Fourv16b_POST);
7816 return;
7817 } else if (VT == MVT::v4i16 || VT == MVT::v4f16 || VT == MVT::v4bf16) {
7818 SelectPostStore(Node, 4, AArch64::ST1Fourv4h_POST);
7819 return;
7820 } else if (VT == MVT::v8i16 || VT == MVT::v8f16 || VT == MVT::v8bf16) {
7821 SelectPostStore(Node, 4, AArch64::ST1Fourv8h_POST);
7822 return;
7823 } else if (VT == MVT::v2i32 || VT == MVT::v2f32) {
7824 SelectPostStore(Node, 4, AArch64::ST1Fourv2s_POST);
7825 return;
7826 } else if (VT == MVT::v4i32 || VT == MVT::v4f32) {
7827 SelectPostStore(Node, 4, AArch64::ST1Fourv4s_POST);
7828 return;
7829 } else if (VT == MVT::v1i64 || VT == MVT::v1f64) {
7830 SelectPostStore(Node, 4, AArch64::ST1Fourv1d_POST);
7831 return;
7832 } else if (VT == MVT::v2i64 || VT == MVT::v2f64) {
7833 SelectPostStore(Node, 4, AArch64::ST1Fourv2d_POST);
7834 return;
7835 }
7836 break;
7837 }
7838 case AArch64ISD::ST2LANEpost: {
7839 VT = Node->getOperand(1).getValueType();
7840 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7841 SelectPostStoreLane(Node, 2, AArch64::ST2i8_POST);
7842 return;
7843 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7844 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7845 SelectPostStoreLane(Node, 2, AArch64::ST2i16_POST);
7846 return;
7847 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7848 VT == MVT::v2f32) {
7849 SelectPostStoreLane(Node, 2, AArch64::ST2i32_POST);
7850 return;
7851 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7852 VT == MVT::v1f64) {
7853 SelectPostStoreLane(Node, 2, AArch64::ST2i64_POST);
7854 return;
7855 }
7856 break;
7857 }
7858 case AArch64ISD::ST3LANEpost: {
7859 VT = Node->getOperand(1).getValueType();
7860 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7861 SelectPostStoreLane(Node, 3, AArch64::ST3i8_POST);
7862 return;
7863 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7864 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7865 SelectPostStoreLane(Node, 3, AArch64::ST3i16_POST);
7866 return;
7867 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7868 VT == MVT::v2f32) {
7869 SelectPostStoreLane(Node, 3, AArch64::ST3i32_POST);
7870 return;
7871 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7872 VT == MVT::v1f64) {
7873 SelectPostStoreLane(Node, 3, AArch64::ST3i64_POST);
7874 return;
7875 }
7876 break;
7877 }
7878 case AArch64ISD::ST4LANEpost: {
7879 VT = Node->getOperand(1).getValueType();
7880 if (VT == MVT::v16i8 || VT == MVT::v8i8) {
7881 SelectPostStoreLane(Node, 4, AArch64::ST4i8_POST);
7882 return;
7883 } else if (VT == MVT::v8i16 || VT == MVT::v4i16 || VT == MVT::v4f16 ||
7884 VT == MVT::v8f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16) {
7885 SelectPostStoreLane(Node, 4, AArch64::ST4i16_POST);
7886 return;
7887 } else if (VT == MVT::v4i32 || VT == MVT::v2i32 || VT == MVT::v4f32 ||
7888 VT == MVT::v2f32) {
7889 SelectPostStoreLane(Node, 4, AArch64::ST4i32_POST);
7890 return;
7891 } else if (VT == MVT::v2i64 || VT == MVT::v1i64 || VT == MVT::v2f64 ||
7892 VT == MVT::v1f64) {
7893 SelectPostStoreLane(Node, 4, AArch64::ST4i64_POST);
7894 return;
7895 }
7896 break;
7897 }
7898 }
7899
7900 // Select the default instruction
7901 SelectCode(Node);
7902}
7903
7904/// createAArch64ISelDag - This pass converts a legalized DAG into a
7905/// AArch64-specific DAG, ready for instruction scheduling.
7907 CodeGenOptLevel OptLevel) {
7908 return new AArch64DAGToDAGISelLegacy(TM, OptLevel);
7909}
7910
7911/// When \p PredVT is a scalable vector predicate in the form
7912/// MVT::nx<M>xi1, it builds the correspondent scalable vector of
7913/// integers MVT::nx<M>xi<bits> s.t. M x bits = 128. When targeting
7914/// structured vectors (NumVec >1), the output data type is
7915/// MVT::nx<M*NumVec>xi<bits> s.t. M x bits = 128. If the input
7916/// PredVT is not in the form MVT::nx<M>xi1, it returns an invalid
7917/// EVT.
7919 unsigned NumVec) {
7920 assert(NumVec > 0 && NumVec < 5 && "Invalid number of vectors.");
7921 if (!PredVT.isScalableVectorOf(MVT::i1))
7922 return EVT();
7923
7924 if (PredVT != MVT::nxv16i1 && PredVT != MVT::nxv8i1 &&
7925 PredVT != MVT::nxv4i1 && PredVT != MVT::nxv2i1)
7926 return EVT();
7927
7928 ElementCount EC = PredVT.getVectorElementCount();
7929 EVT ScalarVT =
7930 EVT::getIntegerVT(Ctx, AArch64::SVEBitsPerBlock / EC.getKnownMinValue());
7931 EVT MemVT = EVT::getVectorVT(Ctx, ScalarVT, EC * NumVec);
7932
7933 return MemVT;
7934}
7935
7936/// Builds an integer vector type large enough to hold \p NumVec instances
7937/// of \p VecVT.
7938static EVT getMultipleVectorType(LLVMContext &Ctx, EVT VecVT, unsigned NumVec) {
7940 VecVT.getVectorElementCount() * NumVec);
7941}
7942
7943/// Return the EVT of the data associated to a memory operation in \p
7944/// Root. If such EVT cannot be retrieved, it returns an invalid EVT.
7946 if (auto *MemIntr = dyn_cast<MemIntrinsicSDNode>(Root))
7947 return MemIntr->getMemoryVT();
7948
7949 if (isa<MemSDNode>(Root)) {
7950 EVT MemVT = cast<MemSDNode>(Root)->getMemoryVT();
7951
7952 EVT DataVT;
7953 if (auto *Load = dyn_cast<LoadSDNode>(Root))
7954 DataVT = Load->getValueType(0);
7955 else if (auto *Load = dyn_cast<MaskedLoadSDNode>(Root))
7956 DataVT = Load->getValueType(0);
7957 else if (auto *Store = dyn_cast<StoreSDNode>(Root))
7958 DataVT = Store->getValue().getValueType();
7959 else if (auto *Store = dyn_cast<MaskedStoreSDNode>(Root))
7960 DataVT = Store->getValue().getValueType();
7961 else
7962 llvm_unreachable("Unexpected MemSDNode!");
7963
7964 return DataVT.changeVectorElementType(Ctx, MemVT.getVectorElementType());
7965 }
7966
7967 const unsigned Opcode = Root->getOpcode();
7968 // For custom ISD nodes, we have to look at them individually to extract the
7969 // type of the data moved to/from memory.
7970 switch (Opcode) {
7971 case AArch64ISD::LD1_MERGE_ZERO:
7972 case AArch64ISD::LD1S_MERGE_ZERO:
7973 case AArch64ISD::LDNF1_MERGE_ZERO:
7974 case AArch64ISD::LDNF1S_MERGE_ZERO:
7975 return cast<VTSDNode>(Root->getOperand(3))->getVT();
7976 case AArch64ISD::ST1_PRED:
7977 return cast<VTSDNode>(Root->getOperand(4))->getVT();
7978 default:
7979 break;
7980 }
7981
7982 if (Opcode != ISD::INTRINSIC_VOID && Opcode != ISD::INTRINSIC_W_CHAIN)
7983 return EVT();
7984
7985 switch (Root->getConstantOperandVal(1)) {
7986 default:
7987 return EVT();
7988 case Intrinsic::aarch64_sme_ldr:
7989 case Intrinsic::aarch64_sme_str:
7990 return MVT::nxv16i8;
7991 case Intrinsic::aarch64_sve_prf:
7992 // We are using an SVE prefetch intrinsic. Type must be inferred from the
7993 // width of the predicate.
7995 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/1);
7996 case Intrinsic::aarch64_sve_ld2_sret:
7997 case Intrinsic::aarch64_sve_ld2q_sret:
7999 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/2);
8000 case Intrinsic::aarch64_sve_st2q:
8002 Ctx, Root->getOperand(4)->getValueType(0), /*NumVec=*/2);
8003 case Intrinsic::aarch64_sve_ld3_sret:
8004 case Intrinsic::aarch64_sve_ld3q_sret:
8006 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/3);
8007 case Intrinsic::aarch64_sve_st3q:
8009 Ctx, Root->getOperand(5)->getValueType(0), /*NumVec=*/3);
8010 case Intrinsic::aarch64_sve_ld4_sret:
8011 case Intrinsic::aarch64_sve_ld4q_sret:
8013 Ctx, Root->getOperand(2)->getValueType(0), /*NumVec=*/4);
8014 case Intrinsic::aarch64_sve_st4q:
8016 Ctx, Root->getOperand(6)->getValueType(0), /*NumVec=*/4);
8017 case Intrinsic::aarch64_sve_ld1_pn_x2:
8018 case Intrinsic::aarch64_sve_ldnt1_pn_x2:
8019 return getMultipleVectorType(Ctx, Root->getValueType(0),
8020 /*NumVec=*/2);
8021 case Intrinsic::aarch64_sve_ld1_pn_x4:
8022 case Intrinsic::aarch64_sve_ldnt1_pn_x4:
8023 return getMultipleVectorType(Ctx, Root->getValueType(0),
8024 /*NumVec=*/4);
8025 case Intrinsic::aarch64_sve_st1_pn_x2:
8026 case Intrinsic::aarch64_sve_stnt1_pn_x2:
8027 return getMultipleVectorType(Ctx, Root->getOperand(2).getValueType(),
8028 /*NumVec=*/2);
8029 case Intrinsic::aarch64_sve_st1_pn_x4:
8030 case Intrinsic::aarch64_sve_stnt1_pn_x4:
8031 return getMultipleVectorType(Ctx, Root->getOperand(2).getValueType(),
8032 /*NumVec=*/4);
8033 case Intrinsic::aarch64_sve_ld1udq:
8034 case Intrinsic::aarch64_sve_st1dq:
8035 return EVT(MVT::nxv1i64);
8036 case Intrinsic::aarch64_sve_ld1uwq:
8037 case Intrinsic::aarch64_sve_st1wq:
8038 return EVT(MVT::nxv1i32);
8039 }
8040}
8041
8042/// SelectAddrModeIndexedSVE - Attempt selection of the addressing mode:
8043/// Base + OffImm * sizeof(MemVT) for Min >= OffImm <= Max
8044/// where Root is the memory access using N for its address.
8045template <int64_t Min, int64_t Max>
8046bool AArch64DAGToDAGISel::SelectAddrModeIndexedSVE(SDNode *Root, SDValue N,
8047 SDValue &Base,
8048 SDValue &OffImm) {
8049 const EVT MemVT = getMemVTFromNode(*(CurDAG->getContext()), Root);
8050 const DataLayout &DL = CurDAG->getDataLayout();
8051 const MachineFrameInfo &MFI = MF->getFrameInfo();
8052
8053 if (N.getOpcode() == ISD::FrameIndex) {
8054 int FI = cast<FrameIndexSDNode>(N)->getIndex();
8055 // We can only encode VL scaled offsets, so only fold in frame indexes
8056 // referencing SVE objects.
8057 if (MFI.hasScalableStackID(FI)) {
8058 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
8059 OffImm = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
8060 return true;
8061 }
8062
8063 return false;
8064 }
8065
8066 if (MemVT == EVT())
8067 return false;
8068
8069 if (N.getOpcode() != ISD::ADD)
8070 return false;
8071
8072 SDValue VScale = N.getOperand(1);
8073 int64_t MulImm = std::numeric_limits<int64_t>::max();
8074 if (VScale.getOpcode() == ISD::VSCALE) {
8075 MulImm = cast<ConstantSDNode>(VScale.getOperand(0))->getSExtValue();
8076 } else if (auto C = dyn_cast<ConstantSDNode>(VScale)) {
8077 int64_t ByteOffset = C->getSExtValue();
8078 const auto KnownVScale =
8080
8081 if (!KnownVScale || ByteOffset % KnownVScale != 0)
8082 return false;
8083
8084 MulImm = ByteOffset / KnownVScale;
8085 } else
8086 return false;
8087
8088 TypeSize TS = MemVT.getSizeInBits();
8089 int64_t MemWidthBytes = static_cast<int64_t>(TS.getKnownMinValue()) / 8;
8090
8091 if ((MulImm % MemWidthBytes) != 0)
8092 return false;
8093
8094 int64_t Offset = MulImm / MemWidthBytes;
8096 return false;
8097
8098 Base = N.getOperand(0);
8099 if (Base.getOpcode() == ISD::FrameIndex) {
8100 int FI = cast<FrameIndexSDNode>(Base)->getIndex();
8101 // We can only encode VL scaled offsets, so only fold in frame indexes
8102 // referencing SVE objects.
8103 if (MFI.hasScalableStackID(FI))
8104 Base = CurDAG->getTargetFrameIndex(FI, TLI->getPointerTy(DL));
8105 }
8106
8107 OffImm = CurDAG->getTargetConstant(Offset, SDLoc(N), MVT::i64);
8108 return true;
8109}
8110
8111/// Select register plus register addressing mode for SVE, with scaled
8112/// offset.
8113bool AArch64DAGToDAGISel::SelectSVERegRegAddrMode(SDValue N, unsigned Scale,
8114 SDValue &Base,
8115 SDValue &Offset) {
8116 if (N.getOpcode() != ISD::ADD)
8117 return false;
8118
8119 // Process an ADD node.
8120 const SDValue LHS = N.getOperand(0);
8121 const SDValue RHS = N.getOperand(1);
8122
8123 // 8 bit data does not come with the SHL node, so it is treated
8124 // separately.
8125 if (Scale == 0) {
8126 Base = LHS;
8127 Offset = RHS;
8128 return true;
8129 }
8130
8131 if (auto C = dyn_cast<ConstantSDNode>(RHS)) {
8132 int64_t ImmOff = C->getSExtValue();
8133 unsigned Size = 1 << Scale;
8134
8135 // To use the reg+reg addressing mode, the immediate must be a multiple of
8136 // the vector element's byte size.
8137 if (ImmOff % Size)
8138 return false;
8139
8140 SDLoc DL(N);
8141 Base = LHS;
8142 Offset = CurDAG->getTargetConstant(ImmOff >> Scale, DL, MVT::i64);
8143 SDValue Ops[] = {Offset};
8144 SDNode *MI = CurDAG->getMachineNode(AArch64::MOVi64imm, DL, MVT::i64, Ops);
8145 Offset = SDValue(MI, 0);
8146 return true;
8147 }
8148
8149 // Check if the RHS is a shift node with a constant.
8150 if (RHS.getOpcode() != ISD::SHL)
8151 return false;
8152
8153 const SDValue ShiftRHS = RHS.getOperand(1);
8154 if (auto *C = dyn_cast<ConstantSDNode>(ShiftRHS))
8155 if (C->getZExtValue() == Scale) {
8156 Base = LHS;
8157 Offset = RHS.getOperand(0);
8158 return true;
8159 }
8160
8161 return false;
8162}
8163
8164bool AArch64DAGToDAGISel::SelectAllActivePredicate(SDValue N) {
8165 const AArch64TargetLowering *TLI =
8166 static_cast<const AArch64TargetLowering *>(getTargetLowering());
8167
8168 return TLI->isAllActivePredicate(*CurDAG, N);
8169}
8170
8171bool AArch64DAGToDAGISel::SelectAnyPredicate(SDValue N) {
8172 return N.getValueType().isScalableVectorOf(MVT::i1);
8173}
8174
8175bool AArch64DAGToDAGISel::SelectSMETileSlice(SDValue N, unsigned MaxSize,
8176 SDValue &Base, SDValue &Offset,
8177 unsigned Scale) {
8178 auto MatchConstantOffset = [&](SDValue CN) -> SDValue {
8179 if (auto *C = dyn_cast<ConstantSDNode>(CN)) {
8180 int64_t ImmOff = C->getSExtValue();
8181 if ((ImmOff > 0 && ImmOff <= MaxSize && (ImmOff % Scale == 0)))
8182 return CurDAG->getTargetConstant(ImmOff / Scale, SDLoc(N), MVT::i64);
8183 }
8184 return SDValue();
8185 };
8186
8187 if (SDValue C = MatchConstantOffset(N)) {
8188 Base = getZeroRegister(*CurDAG, SDLoc(N), MVT::i32);
8189 Offset = C;
8190 return true;
8191 }
8192
8193 // Try to untangle an ADD node into a 'reg + offset'
8194 if (CurDAG->isBaseWithConstantOffset(N)) {
8195 if (SDValue C = MatchConstantOffset(N.getOperand(1))) {
8196 Base = N.getOperand(0);
8197 Offset = C;
8198 return true;
8199 }
8200 }
8201
8202 // By default, just match reg + 0.
8203 Base = N;
8204 Offset = CurDAG->getTargetConstant(0, SDLoc(N), MVT::i64);
8205 return true;
8206}
8207
8208bool AArch64DAGToDAGISel::SelectCmpBranchUImm6Operand(SDNode *P, SDValue N,
8209 SDValue &Imm) {
8211 static_cast<AArch64CC::CondCode>(P->getConstantOperandVal(1));
8212 if (auto *CN = dyn_cast<ConstantSDNode>(N)) {
8213 // Check conservatively if the immediate fits the valid range [0, 64).
8214 // Immediate variants for GE and HS definitely need to be decremented
8215 // when lowering the pseudos later, so an immediate of 1 would become 0.
8216 // For the inverse conditions LT and LO we don't know for sure if they
8217 // will need a decrement but should the decision be made to reverse the
8218 // branch condition, we again end up with the need to decrement.
8219 // The same argument holds for LE, LS, GT and HI and possibly
8220 // incremented immediates. This can lead to slightly less optimal
8221 // codegen, e.g. we never codegen the legal case
8222 // cblt w0, #63, A
8223 // because we could end up with the illegal case
8224 // cbge w0, #64, B
8225 // should the decision to reverse the branch direction be made. For the
8226 // lower bound cases this is no problem since we can express comparisons
8227 // against 0 with either tbz/tnbz or using wzr/xzr.
8228 uint64_t LowerBound = 0, UpperBound = 64;
8229 switch (CC) {
8230 case AArch64CC::GE:
8231 case AArch64CC::HS:
8232 case AArch64CC::LT:
8233 case AArch64CC::LO:
8234 LowerBound = 1;
8235 break;
8236 case AArch64CC::LE:
8237 case AArch64CC::LS:
8238 case AArch64CC::GT:
8239 case AArch64CC::HI:
8240 UpperBound = 63;
8241 break;
8242 default:
8243 break;
8244 }
8245
8246 if (CN->getAPIntValue().uge(LowerBound) &&
8247 CN->getAPIntValue().ult(UpperBound)) {
8248 SDLoc DL(N);
8249 Imm = CurDAG->getTargetConstant(CN->getZExtValue(), DL, N.getValueType());
8250 return true;
8251 }
8252 }
8253
8254 return false;
8255}
8256
8257template <bool MatchCBB>
8258bool AArch64DAGToDAGISel::SelectCmpBranchExtOperand(SDValue N, SDValue &Reg,
8259 SDValue &ExtType) {
8260
8261 // Use an invalid shift-extend value to indicate we don't need to extend later
8262 if (N.getOpcode() == ISD::AssertZext || N.getOpcode() == ISD::AssertSext) {
8263 EVT Ty = cast<VTSDNode>(N.getOperand(1))->getVT();
8264 if (Ty != (MatchCBB ? MVT::i8 : MVT::i16))
8265 return false;
8266 Reg = N.getOperand(0);
8267 ExtType = CurDAG->getSignedTargetConstant(AArch64_AM::InvalidShiftExtend,
8268 SDLoc(N), MVT::i32);
8269 return true;
8270 }
8271
8273
8274 if ((MatchCBB && (ET == AArch64_AM::UXTB || ET == AArch64_AM::SXTB)) ||
8275 (!MatchCBB && (ET == AArch64_AM::UXTH || ET == AArch64_AM::SXTH))) {
8276 Reg = N.getOperand(0);
8277 ExtType =
8278 CurDAG->getTargetConstant(getExtendEncoding(ET), SDLoc(N), MVT::i32);
8279 return true;
8280 }
8281
8282 return false;
8283}
8284
8285/// Try to fold AArch64 CSEL/FCMP patterns to FMAXNM/FMINNM.
8286///
8287/// This is intentionally done in PreprocessISelDAG rather than DAGCombine:
8288/// doing this earlier based on the defining operation of X can be invalidated
8289/// by later DAG combines. At this point the DAG is being prepared for
8290/// instruction selection, so the use of isKnownNeverSNaN(X) applies to the
8291/// final SDValue being selected.
8292/// Only handles FCMP(X, C) with scalar FP types, where C is a non-NaN constant.
8293/// The nsz requirement is needed only when C is zero, to avoid signed-zero
8294/// mismatches. The never-sNaN check is required because AArch64 FMAXNM/FMINNM
8295/// differ from fcmp+fcsel for signaling NaN inputs.
8296bool AArch64DAGToDAGISel::tryFoldCselToFMaxMin(SDNode *N) {
8297 EVT VT = N->getValueType(0);
8298
8299 // Scalar FP only.
8300 if (!VT.isFloatingPoint() || VT.isVector())
8301 return false;
8302
8303 SDValue TVal = N->getOperand(0);
8304 SDValue FVal = N->getOperand(1);
8305 SDValue CCVal = N->getOperand(2);
8306 SDValue Cmp = N->getOperand(3);
8307
8308 if (Cmp.getOpcode() != AArch64ISD::FCMP)
8309 return false;
8310
8311 auto *CC = dyn_cast<ConstantSDNode>(CCVal);
8312 if (!CC)
8313 return false;
8314
8315 SDValue CmpLHS = Cmp.getOperand(0);
8316 SDValue CmpRHS = Cmp.getOperand(1);
8317 unsigned CondCode = CC->getZExtValue();
8318
8319 // Map VT and operation (max/min) to machine opcode.
8320 auto getOpc = [](EVT VT, bool isMax) -> unsigned {
8321 if (VT == MVT::f16)
8322 return isMax ? AArch64::FMAXNMHrr : AArch64::FMINNMHrr;
8323 else if (VT == MVT::f32)
8324 return isMax ? AArch64::FMAXNMSrr : AArch64::FMINNMSrr;
8325 else if (VT == MVT::f64)
8326 return isMax ? AArch64::FMAXNMDrr : AArch64::FMINNMDrr;
8327 else
8328 return 0; // unsupported
8329 };
8330
8331 // Determine whether to use max or min based on condition code and operands.
8332 bool isMax;
8333 if (CondCode == AArch64CC::GT || CondCode == AArch64CC::GE) {
8334 if (TVal == CmpLHS && FVal == CmpRHS)
8335 isMax = true;
8336 else
8337 return false;
8338 } else if (CondCode == AArch64CC::MI || CondCode == AArch64CC::LS) {
8339 if (TVal == CmpLHS && FVal == CmpRHS)
8340 isMax = false;
8341 else
8342 return false;
8343 } else {
8344 return false;
8345 }
8346
8347 // Get the machine opcode for this VT and operation.
8348 unsigned Opc = getOpc(VT, isMax);
8349 if (!Opc)
8350 return false;
8351
8352 // Constant must be non-NaN.
8353 auto *CFP = dyn_cast<ConstantFPSDNode>(CmpRHS);
8354 if (!CFP || CFP->getValueAPF().isNaN())
8355 return false;
8356
8357 // nsz flag required only when constant is zero: fmaxnm(+0,-0)=+0 differs from
8358 // fcmp+select's -0. For non-zero constants, semantics are identical.
8359 if (CFP->isZero() && !N->getFlags().hasNoSignedZeros())
8360 return false;
8361
8362 // Only fold if variable operand is never sNaN.
8363 // This runs after DAG combines, so later combines cannot remove a defining
8364 // operation used by isKnownNeverSNaN().
8365 if (!CurDAG->isKnownNeverSNaN(CmpLHS))
8366 return false;
8367
8368 CurDAG->SelectNodeTo(N, Opc, VT, CmpLHS, CmpRHS);
8369 return true;
8370}
8371
8372void AArch64DAGToDAGISel::PreprocessISelDAG() {
8373 bool MadeChange = false;
8374 for (SDNode &N : llvm::make_early_inc_range(CurDAG->allnodes())) {
8375 if (N.use_empty())
8376 continue;
8377
8378 SDValue Result;
8379 switch (N.getOpcode()) {
8380 case ISD::SCALAR_TO_VECTOR: {
8381 EVT ScalarTy = N.getValueType(0).getVectorElementType();
8382 if ((ScalarTy == MVT::i32 || ScalarTy == MVT::i64) &&
8383 ScalarTy == N.getOperand(0).getValueType())
8384 Result = addBitcastHints(*CurDAG, N);
8385
8386 break;
8387 }
8388 case AArch64ISD::VSHL: {
8389 // Undo mul(shl(A,C),B) -> shl(mul(A,B),C) canonicalisation when A is an
8390 // extend that can be folded into the shift.
8391 EVT VT = N.getValueType(0);
8392 SDValue A, B, C = N.getOperand(1);
8393 if (sd_match(N.getOperand(0),
8395 m_SExt(m_Value()))),
8396 m_Value(B))))) {
8397 // If both mul operands are extended, preserve the smull/umull idiom.
8398 if (B.getOpcode() == A.getOpcode())
8399 break;
8400 SDLoc DL(&N);
8401 SDValue SHL = CurDAG->getNode(AArch64ISD::VSHL, DL, VT, A, C);
8402 Result = CurDAG->getNode(ISD::MUL, DL, VT, SHL, B);
8403 }
8404 break;
8405 }
8406 default:
8407 break;
8408 }
8409
8410 if (Result) {
8411 LLVM_DEBUG(dbgs() << "AArch64 DAG preprocessing replacing:\nOld: ");
8412 LLVM_DEBUG(N.dump(CurDAG));
8413 LLVM_DEBUG(dbgs() << "\nNew: ");
8414 LLVM_DEBUG(Result.dump(CurDAG));
8415 LLVM_DEBUG(dbgs() << "\n");
8416
8417 CurDAG->ReplaceAllUsesOfValueWith(SDValue(&N, 0), Result);
8418 MadeChange = true;
8419 }
8420 }
8421
8422 if (MadeChange)
8423 CurDAG->RemoveDeadNodes();
8424
8426}
static std::optional< APInt > GetNEONSplatValue(SDValue N, const AArch64Subtarget *Subtarget)
static SDValue Widen(SelectionDAG *CurDAG, SDValue N)
static bool isBitfieldExtractOpFromSExtInReg(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms)
static int getIntOperandFromRegisterString(StringRef RegString)
static SDValue NarrowVector(SDValue V128Reg, SelectionDAG &DAG)
NarrowVector - Given a value in the V128 register class, produce the equivalent value in the V64 regi...
static bool isBitfieldDstMask(uint64_t DstMask, const APInt &BitsToBeInserted, unsigned NumberOfIgnoredHighBits, EVT VT)
Does DstMask form a complementary pair with the mask provided by BitsToBeInserted,...
static SDValue narrowIfNeeded(SelectionDAG *CurDAG, SDValue N)
Instructions that accept extend modifiers like UXTW expect the register being extended to be a GPR32,...
static bool isSeveralBitsPositioningOpFromShl(const uint64_t ShlImm, SDValue Op, SDValue &Src, int &DstLSB, int &Width)
static bool isBitfieldPositioningOp(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, SDValue &Src, int &DstLSB, int &Width)
Does this tree qualify as an attempt to move a bitfield into position, essentially "(and (shl VAL,...
static bool isOpcWithIntImmediate(const SDNode *N, unsigned Opc, uint64_t &Imm)
static bool tryBitfieldInsertOpFromOrAndImm(SDNode *N, SelectionDAG *CurDAG)
static std::tuple< SDValue, SDValue > extractPtrauthBlendDiscriminators(SDValue Disc, SelectionDAG *DAG)
static SDValue addBitcastHints(SelectionDAG &DAG, SDNode &N)
addBitcastHints - This method adds bitcast hints to the operands of a node to help instruction select...
static void getUsefulBitsFromOrWithShiftedReg(SDValue Op, APInt &UsefulBits, unsigned Depth)
static bool isBitfieldExtractOpFromAnd(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &LSB, unsigned &MSB, unsigned NumberOfIgnoredLowBits, bool BiggerPattern)
static bool isBitfieldExtractOp(SelectionDAG *CurDAG, SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms, unsigned NumberOfIgnoredLowBits=0, bool BiggerPattern=false)
static bool isShiftedMask(uint64_t Mask, EVT VT)
bool SelectSMETile(unsigned &BaseReg, unsigned TileNum)
static EVT getMemVTFromNode(LLVMContext &Ctx, SDNode *Root)
Return the EVT of the data associated to a memory operation in Root.
static bool checkCVTFixedPointOperandWithFBits(SelectionDAG *CurDAG, SDValue N, SDValue &FixedPos, unsigned RegWidth, bool isReciprocal)
static bool isWorthFoldingADDlow(SDValue N)
If there's a use of this ADDlow that's not itself a load/store then we'll need to create a real ADD i...
static AArch64_AM::ShiftExtendType getShiftTypeForNode(SDValue N)
getShiftTypeForNode - Translate a shift node to the corresponding ShiftType value.
static bool isSeveralBitsExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &LSB, unsigned &MSB)
static unsigned SelectOpcodeFromVT(EVT VT, ArrayRef< unsigned > Opcodes)
This function selects an opcode from a list of opcodes, which is expected to be the opcode for { 8-bi...
static EVT getPackedVectorTypeFromPredicateType(LLVMContext &Ctx, EVT PredVT, unsigned NumVec)
When PredVT is a scalable vector predicate in the form MVT::nx<M>xi1, it builds the correspondent sca...
static bool checkCVTFixedPointOperandWithFBitsForVectors(SelectionDAG *CurDAG, SDValue N, SDValue &FixedPos, unsigned RegWidth, bool isReciprocal)
static SDValue getZeroRegister(SelectionDAG &DAG, SDLoc DL, EVT VT)
Returns a copy from WZR or XZR.
static bool isPreferredADD(int64_t ImmOff)
static void getUsefulBitsFromBitfieldMoveOpd(SDValue Op, APInt &UsefulBits, uint64_t Imm, uint64_t MSB, unsigned Depth)
static SDValue getLeftShift(SelectionDAG *CurDAG, SDValue Op, int ShlAmount)
Create a machine node performing a notional SHL of Op by ShlAmount.
static bool isWorthFoldingSHL(SDValue V)
Determine whether it is worth it to fold SHL into the addressing mode.
static bool isBitfieldPositioningOpFromAnd(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, const uint64_t NonZeroBits, SDValue &Src, int &DstLSB, int &Width)
static void getUsefulBitsFromBFM(SDValue Op, SDValue Orig, APInt &UsefulBits, unsigned Depth)
static bool isBitfieldExtractOpFromShr(SDNode *N, unsigned &Opc, SDValue &Opd0, unsigned &Immr, unsigned &Imms, bool BiggerPattern)
static bool tryOrrWithShift(SDNode *N, SDValue OrOpd0, SDValue OrOpd1, SDValue Src, SDValue Dst, SelectionDAG *CurDAG, const bool BiggerPattern)
static void getUsefulBitsForUse(SDNode *UserNode, APInt &UsefulBits, SDValue Orig, unsigned Depth)
static bool isMemOpOrPrefetch(SDNode *N)
static void getUsefulBitsFromUBFM(SDValue Op, APInt &UsefulBits, unsigned Depth)
static bool tryBitfieldInsertOpFromOr(SDNode *N, const APInt &UsefulBits, SelectionDAG *CurDAG)
static APInt DecodeFMOVImm(uint64_t Imm, unsigned RegWidth)
static void getUsefulBitsFromAndWithImmediate(SDValue Op, APInt &UsefulBits, unsigned Depth)
static std::optional< APInt > DecodeNEONSplat(SDValue N, const AArch64Subtarget *Subtarget)
static void getUsefulBits(SDValue Op, APInt &UsefulBits, unsigned Depth=0)
static bool isIntImmediateEq(SDValue N, const uint64_t ImmExpected)
static EVT getMultipleVectorType(LLVMContext &Ctx, EVT VecVT, unsigned NumVec)
Builds an integer vector type large enough to hold NumVec instances of VecVT.
static AArch64_AM::ShiftExtendType getExtendTypeForNode(SDValue N, bool IsLoadStore=false)
getExtendTypeForNode - Translate an extend node to the corresponding ExtendType value.
static bool isIntImmediate(const SDNode *N, uint64_t &Imm)
isIntImmediate - This method tests to see if the node is a constant operand.
static bool isWorthFoldingIntoOrrWithShift(SDValue Dst, SelectionDAG *CurDAG, SDValue &ShiftedOperand, uint64_t &EncodedShiftImm)
static bool isValidAsScaledImmediate(int64_t Offset, unsigned Range, unsigned Size)
Check if the immediate offset is valid as a scaled immediate.
static bool isBitfieldPositioningOpFromShl(SelectionDAG *CurDAG, SDValue Op, bool BiggerPattern, const uint64_t NonZeroBits, SDValue &Src, int &DstLSB, int &Width)
static SDValue WidenVector(SDValue V64Reg, SelectionDAG &DAG)
WidenVector - Given a value in the V64 register class, produce the equivalent value in the V128 regis...
static Register createDTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of D-registers using the registers in Regs.
static Register createQTuple(ArrayRef< Register > Regs, MachineIRBuilder &MIB)
Create a tuple of Q-registers using the registers in Regs.
static Register createTuple(ArrayRef< Register > Regs, const unsigned RegClassIDs[], const unsigned SubRegs[], MachineIRBuilder &MIB)
Create a REG_SEQUENCE instruction using the registers in Regs.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
This file implements the APSInt class, which is a simple class that represents an arbitrary sized int...
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
dxil translate DXIL Translate Metadata
#define DEBUG_TYPE
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
std::pair< Instruction::BinaryOps, Value * > OffsetOp
Find all possible pairs (BinOp, RHS) that BinOp V, RHS can be simplified.
static constexpr Value * getValue(Ty &ValueOrUse)
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define I(x, y, z)
Definition MD5.cpp:57
Register Reg
#define R2(n)
Promote Memory to Register
Definition Mem2Reg.cpp:110
ConstantRange Range(APInt(BitWidth, Low), APInt(BitWidth, High))
uint64_t High
OptimizedStructLayoutField Field
#define P(N)
#define INITIALIZE_PASS(passName, arg, name, cfg, analysis)
Definition PassSupport.h:56
Contains matchers for matching SelectionDAG nodes and values.
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
#define LLVM_DEBUG(...)
Definition Debug.h:119
#define PASS_NAME
Value * RHS
Value * LHS
AArch64DAGToDAGISelPass(AArch64TargetMachine &TM)
const AArch64InstrInfo * getInstrInfo() const override
const AArch64TargetLowering * getTargetLowering() const override
bool isStreaming() const
Returns true if the function has a streaming body.
bool isX16X17Safer() const
Returns whether the operating system makes it safer to store sensitive values in x16 and x17 as oppos...
unsigned getSVEVectorSizeInBits() const
bool isAllActivePredicate(const SelectionDAG &DAG, SDValue N) const
Register matchRegisterName(StringRef RegName) const
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
unsigned popcount() const
Count the number of bits set.
Definition APInt.h:1690
LLVM_ABI APInt zextOrTrunc(unsigned width) const
Zero extend or truncate to width.
Definition APInt.cpp:1078
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned countl_zero() const
The APInt version of std::countl_zero.
Definition APInt.h:1618
static LLVM_ABI APInt getSplat(unsigned NewLen, const APInt &V)
Return a value containing V broadcasted over NewLen bits.
Definition APInt.cpp:648
void flipAllBits()
Toggle every bit to its opposite value.
Definition APInt.h:1472
bool isShiftedMask() const
Return true if this APInt value contains a non-empty sequence of ones with the remainder zero.
Definition APInt.h:506
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
void lshrInPlace(unsigned ShiftAmt)
Logical right-shift this APInt by ShiftAmt in place.
Definition APInt.h:860
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
iterator begin() const
Definition ArrayRef.h:129
const Constant * getConstVal() const
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
const GlobalValue * getGlobal() const
const TargetRegisterClass * getInlineAsmMemoryOperandRegClass(InlineAsm::ConstraintCode C) const override
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
This class is used to represent ISD::LOAD nodes.
unsigned getID() const
getID() - Return the register class ID number.
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
unsigned getNumOperands() const
Return number of MDNode operands.
Definition Metadata.h:1443
bool equalsStr(StringRef Str) const
Definition Metadata.h:924
Metadata * get() const
Definition Metadata.h:931
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
static MVT getVectorVT(MVT VT, unsigned NumElements)
bool hasScalableStackID(int ObjectIdx) const
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
A description of a memory reference used in the backend.
const MDNode * getMemCacheHint() const
Return the cache hint metadata for the memory reference.
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
iterator_range< user_iterator > users()
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
const SDValue & getOperand(unsigned i) const
uint64_t getConstantOperandVal(unsigned i) const
unsigned getOpcode() const
SelectionDAGISelPass(std::unique_ptr< SelectionDAGISel > Selector)
SelectionDAGISel - This is the common base class used for SelectionDAG-based pattern-matching instruc...
virtual void PreprocessISelDAG()
PreprocessISelDAG - This hook allows targets to hack on the graph before instruction selection starts...
virtual bool runOnMachineFunction(MachineFunction &mf)
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDNode * SelectNodeTo(SDNode *N, unsigned MachineOpc, EVT VT)
These are used for target selectors to mutate the specified node to have the specified return type,...
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
static constexpr unsigned MaxRecursionDepth
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetInsertSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand, SDValue Subreg)
A convenience function for creating TargetInstrInfo::INSERT_SUBREG nodes.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
void reserve(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
std::pair< StringRef, StringRef > split(char Separator) const
Split into two substrings around the first occurrence of a separator character.
Definition StringRef.h:736
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
LLVM Value Representation.
Definition Value.h:75
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI Align getPointerAlignment(const DataLayout &DL) const
Returns an alignment of the pointer value.
Definition Value.cpp:1002
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
uint32_t parseGenericRegister(StringRef Name)
static uint64_t decodeLogicalImmediate(uint64_t val, unsigned regSize)
decodeLogicalImmediate - Decode a logical immediate value in the form "N:immr:imms" (where the immr a...
static unsigned getShiftValue(unsigned Imm)
getShiftValue - Extract the shift value.
static bool isLogicalImmediate(uint64_t imm, unsigned regSize)
isLogicalImmediate - Return true if the immediate is valid for a logical immediate instruction of the...
static uint64_t decodeAdvSIMDModImmType12(uint8_t Imm)
constexpr bool isLegalArithImmed(const uint64_t C)
isLegalArithImmed -
static uint64_t decodeAdvSIMDModImmType11(uint8_t Imm)
unsigned getExtendEncoding(AArch64_AM::ShiftExtendType ET)
Mapping from extend bits to required operation: shifter: 000 ==> uxtb 001 ==> uxth 010 ==> uxtw 011 =...
static uint64_t decodeAdvSIMDModImmType10(uint8_t Imm)
static bool isSVELogicalImm(unsigned SizeInBits, uint64_t ImmVal, uint64_t &Encoding)
constexpr unsigned getArithImmedShift(const uint64_t C)
getArithImmedShift - assumes C is a legal immediate for arithmetic instructions and
static bool isSVECpyDupImm(int SizeInBits, int64_t Val, int32_t &Imm, int32_t &Shift)
static AArch64_AM::ShiftExtendType getShiftType(unsigned Imm)
getShiftType - Extract the shift type.
static unsigned getShifterImm(AArch64_AM::ShiftExtendType ST, unsigned Imm)
getShifterImm - Encode the shift type and amount: imm: 6-bit shift amount shifter: 000 ==> lsl 001 ==...
static bool isSignExtendShiftType(AArch64_AM::ShiftExtendType Type)
isSignExtendShiftType - Returns true if Type is sign extending.
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
static constexpr unsigned SVEBitsPerBlock
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:682
@ AssertAlign
AssertAlign - These nodes record if a register contains a value that has a known alignment and the tr...
Definition ISDOpcodes.h:71
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ READ_REGISTER
READ_REGISTER, WRITE_REGISTER - This node represents llvm.register on the DAG, which implements the n...
Definition ISDOpcodes.h:141
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ VSCALE
VSCALE(IMM) - Returns the runtime scaling factor used to calculate the number of elements within a sc...
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ FREEZE
FREEZE - FREEZE(VAL) returns an arbitrary value if VAL is UNDEF (or is evaluated to UNDEF),...
Definition ISDOpcodes.h:243
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
MemIndexedMode
MemIndexedMode enum - This enum defines the load / store indexed addressing modes.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
BinaryOpc_match< LHS, RHS, true > m_Mul(const LHS &L, const RHS &R)
Or< Preds... > m_AnyOf(const Preds &...preds)
auto m_SExt(const Opnd &Op)
bool sd_match(SDValue N, Pattern &&P)
UnaryOpc_match< Opnd > m_ZExt(const Opnd &Op)
Value_match m_Value()
Match any valid SDValue.
NUses_match< 1, Value_match > m_OneUse()
Not(const Pred &P) -> Not< Pred >
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
@ Offset
Definition DWP.cpp:577
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
unsigned CheckFixedPointOperandConstant(APFloat &FVal, unsigned RegWidth, bool isReciprocal)
@ Known
Known to have no common set bits.
@ Undef
Value of the register doesn't matter.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool isStrongerThanMonotonic(AtomicOrdering AO)
int countr_one(T Value)
Count the number of ones from the least significant bit to the first zero bit.
Definition bit.h:315
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
iterator_range< early_inc_iterator_impl< detail::IterOfRange< RangeT > > > make_early_inc_range(RangeT &&Range)
Make a range that does early increment to allow mutation of the underlying range without disrupting i...
Definition STLExtras.h:649
constexpr bool isShiftedMask_32(uint32_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (32 bit ver...
Definition MathExtras.h:268
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
OutputIt transform(R &&Range, OutputIt d_first, UnaryFunction F)
Wrapper function around std::transform to apply a function to a range and store the result elsewhere.
Definition STLExtras.h:2042
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr bool isMask_64(uint64_t Value)
Return true if the argument is a non-empty sequence of ones starting at the least significant bit wit...
Definition MathExtras.h:262
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
CodeGenOptLevel
Code generation optimization level.
Definition CodeGen.h:227
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ATTRIBUTE_VISIBILITY_DEFAULT AnalysisKey InnerAnalysisManagerProxy< AnalysisManagerT, IRUnitT, ExtraArgTs... >::Key
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
FunctionPass * createAArch64ISelDag(AArch64TargetMachine &TM, CodeGenOptLevel OptLevel)
createAArch64ISelDag - This pass converts a legalized DAG into a AArch64-specific DAG,...
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isNullFPConstant(SDValue V)
Returns true if V is an FP constant with a value of positive zero.
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
AArch64MemoryHint toAArch64MemoryHint(Int I)
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
Extended Value Type.
Definition ValueTypes.h:35
bool isScalableVectorOf(EVT EltVT) const
Return true if this is a scalable vector with matching element type.
Definition ValueTypes.h:192
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
EVT getDoubleNumVectorElementsVT(LLVMContext &Context) const
Definition ValueTypes.h:494
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
unsigned getVectorMinNumElements() const
Given a vector type, return the minimum number of elements it contains.
Definition ValueTypes.h:382
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isFixedLengthVector() const
Definition ValueTypes.h:199
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool isScalableVector() const
Return true if this is a vector type where the runtime length is machine dependent.
Definition ValueTypes.h:187
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool is64BitVector() const
Return true if this is a 64-bit vector type.
Definition ValueTypes.h:225
Matching combinators.