LLVM 24.0.0git
AArch64PostLegalizerLowering.cpp
Go to the documentation of this file.
1//=== AArch64PostLegalizerLowering.cpp --------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// Post-legalization lowering for instructions.
11///
12/// This is used to offload pattern matching from the selector.
13///
14/// For example, this combiner will notice that a G_SHUFFLE_VECTOR is actually
15/// a G_ZIP, G_UZP, etc.
16///
17/// General optimization combines should be handled by either the
18/// AArch64PostLegalizerCombiner or the AArch64PreLegalizerCombiner.
19///
20//===----------------------------------------------------------------------===//
21
22#include "AArch64.h"
23#include "AArch64ExpandImm.h"
26#include "AArch64Subtarget.h"
47#include "llvm/IR/InstrTypes.h"
49#include <optional>
50
51#define GET_GICOMBINER_DEPS
52#include "AArch64GenPostLegalizeGILowering.inc"
53#undef GET_GICOMBINER_DEPS
54
55#define DEBUG_TYPE "aarch64-postlegalizer-lowering"
56
57using namespace llvm;
58using namespace MIPatternMatch;
59using namespace AArch64GISelUtils;
60
61#define GET_GICOMBINER_TYPES
62#include "AArch64GenPostLegalizeGILowering.inc"
63#undef GET_GICOMBINER_TYPES
64
65namespace {
66
67/// Represents a pseudo instruction which replaces a G_SHUFFLE_VECTOR.
68///
69/// Used for matching target-supported shuffles before codegen.
70struct ShuffleVectorPseudo {
71 unsigned Opc; ///< Opcode for the instruction. (E.g. G_ZIP1)
72 Register Dst; ///< Destination register.
73 SmallVector<SrcOp, 2> SrcOps; ///< Source registers.
74 ShuffleVectorPseudo(unsigned Opc, Register Dst,
75 std::initializer_list<SrcOp> SrcOps)
76 : Opc(Opc), Dst(Dst), SrcOps(SrcOps){};
77 ShuffleVectorPseudo() = default;
78};
79
80/// Check if a G_EXT instruction can handle a shuffle mask \p M when the vector
81/// sources of the shuffle are different.
82std::optional<std::pair<bool, uint64_t>> getExtMask(ArrayRef<int> M,
83 unsigned NumElts) {
84 // Look for the first non-undef element.
85 auto FirstRealElt = find_if(M, [](int Elt) { return Elt >= 0; });
86 if (FirstRealElt == M.end())
87 return std::nullopt;
88
89 // Use APInt to handle overflow when calculating expected element.
90 unsigned MaskBits = APInt(32, NumElts * 2).logBase2();
91 APInt ExpectedElt = APInt(MaskBits, *FirstRealElt + 1, false, true);
92
93 // The following shuffle indices must be the successive elements after the
94 // first real element.
95 if (any_of(
96 make_range(std::next(FirstRealElt), M.end()),
97 [&ExpectedElt](int Elt) { return Elt != ExpectedElt++ && Elt >= 0; }))
98 return std::nullopt;
99
100 // The index of an EXT is the first element if it is not UNDEF.
101 // Watch out for the beginning UNDEFs. The EXT index should be the expected
102 // value of the first element. E.g.
103 // <-1, -1, 3, ...> is treated as <1, 2, 3, ...>.
104 // <-1, -1, 0, 1, ...> is treated as <2*NumElts-2, 2*NumElts-1, 0, 1, ...>.
105 // ExpectedElt is the last mask index plus 1.
106 uint64_t Imm = ExpectedElt.getZExtValue();
107 bool ReverseExt = false;
108
109 // There are two difference cases requiring to reverse input vectors.
110 // For example, for vector <4 x i32> we have the following cases,
111 // Case 1: shufflevector(<4 x i32>,<4 x i32>,<-1, -1, -1, 0>)
112 // Case 2: shufflevector(<4 x i32>,<4 x i32>,<-1, -1, 7, 0>)
113 // For both cases, we finally use mask <5, 6, 7, 0>, which requires
114 // to reverse two input vectors.
115 if (Imm < NumElts)
116 ReverseExt = true;
117 else
118 Imm -= NumElts;
119 return std::make_pair(ReverseExt, Imm);
120}
121
122/// Helper function for matchINS.
123///
124/// \returns a value when \p M is an ins mask for \p NumInputElements.
125///
126/// First element of the returned pair is true when the produced
127/// G_INSERT_VECTOR_ELT destination should be the LHS of the G_SHUFFLE_VECTOR.
128///
129/// Second element is the destination lane for the G_INSERT_VECTOR_ELT.
130std::optional<std::pair<bool, int>> isINSMask(ArrayRef<int> M,
131 int NumInputElements) {
132 if (M.size() != static_cast<size_t>(NumInputElements))
133 return std::nullopt;
134 int NumLHSMatch = 0, NumRHSMatch = 0;
135 int LastLHSMismatch = -1, LastRHSMismatch = -1;
136 for (int Idx = 0; Idx < NumInputElements; ++Idx) {
137 if (M[Idx] == -1) {
138 ++NumLHSMatch;
139 ++NumRHSMatch;
140 continue;
141 }
142 M[Idx] == Idx ? ++NumLHSMatch : LastLHSMismatch = Idx;
143 M[Idx] == Idx + NumInputElements ? ++NumRHSMatch : LastRHSMismatch = Idx;
144 }
145 const int NumNeededToMatch = NumInputElements - 1;
146 if (NumLHSMatch == NumNeededToMatch)
147 return std::make_pair(true, LastLHSMismatch);
148 if (NumRHSMatch == NumNeededToMatch)
149 return std::make_pair(false, LastRHSMismatch);
150 return std::nullopt;
151}
152
153/// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with a
154/// G_REV instruction. Returns the appropriate G_REV opcode in \p Opc.
155bool matchREV(MachineInstr &MI, MachineRegisterInfo &MRI,
156 ShuffleVectorPseudo &MatchInfo) {
157 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
158 ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
159 Register Dst = MI.getOperand(0).getReg();
160 Register Src = MI.getOperand(1).getReg();
161 LLT Ty = MRI.getType(Dst);
162 unsigned EltSize = Ty.getScalarSizeInBits();
163
164 // Element size for a rev cannot be 64.
165 if (EltSize == 64)
166 return false;
167
168 unsigned NumElts = Ty.getNumElements();
169
170 // Try to produce a G_REV instruction
171 for (unsigned LaneSize : {64U, 32U, 16U}) {
172 if (isREVMask(ShuffleMask, EltSize, NumElts, LaneSize)) {
173 unsigned Opcode;
174 if (LaneSize == 64U)
175 Opcode = AArch64::G_REV64;
176 else if (LaneSize == 32U)
177 Opcode = AArch64::G_REV32;
178 else
179 Opcode = AArch64::G_BSWAP;
180
181 MatchInfo = ShuffleVectorPseudo(Opcode, Dst, {Src});
182 return true;
183 }
184 }
185
186 return false;
187}
188
189/// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with
190/// a G_TRN1 or G_TRN2 instruction.
191bool matchTRN(MachineInstr &MI, MachineRegisterInfo &MRI,
192 ShuffleVectorPseudo &MatchInfo) {
193 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
194 unsigned WhichResult;
195 unsigned OperandOrder = 0;
196 ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
197 Register Dst = MI.getOperand(0).getReg();
198 unsigned NumElts = MRI.getType(Dst).getNumElements();
199 bool TRNMask = isTRNMask(ShuffleMask, NumElts, WhichResult, OperandOrder);
200 if (!TRNMask && !isTRN_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
201 return false;
202 unsigned Opc = (WhichResult == 0) ? AArch64::G_TRN1 : AArch64::G_TRN2;
203 Register V1 = MI.getOperand(OperandOrder == 0 ? 1 : 2).getReg();
204 Register V2 = MI.getOperand(OperandOrder == 0 && TRNMask ? 2 : 1).getReg();
205 MatchInfo = ShuffleVectorPseudo(Opc, Dst, {V1, V2});
206 return true;
207}
208
209/// \return true if a G_SHUFFLE_VECTOR instruction \p MI can be replaced with
210/// a G_UZP1 or G_UZP2 instruction.
211///
212/// \param [in] MI - The shuffle vector instruction.
213/// \param [out] MatchInfo - Either G_UZP1 or G_UZP2 on success.
214bool matchUZP(MachineInstr &MI, MachineRegisterInfo &MRI,
215 ShuffleVectorPseudo &MatchInfo) {
216 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
217 unsigned WhichResult;
218 ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
219 Register Dst = MI.getOperand(0).getReg();
220 unsigned NumElts = MRI.getType(Dst).getNumElements();
221 bool UZPMask = isUZPMask(ShuffleMask, NumElts, WhichResult);
222 if (!UZPMask && !isUZP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
223 return false;
224 unsigned Opc = (WhichResult == 0) ? AArch64::G_UZP1 : AArch64::G_UZP2;
225 Register V1 = MI.getOperand(1).getReg();
226 Register V2 = MI.getOperand(UZPMask ? 2 : 1).getReg();
227 MatchInfo = ShuffleVectorPseudo(Opc, Dst, {V1, V2});
228 return true;
229}
230
231bool matchZip(MachineInstr &MI, MachineRegisterInfo &MRI,
232 ShuffleVectorPseudo &MatchInfo) {
233 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
234 unsigned WhichResult;
235 unsigned OperandOrder = 0;
236 ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
237 Register Dst = MI.getOperand(0).getReg();
238 unsigned NumElts = MRI.getType(Dst).getNumElements();
239 bool ZIPMask = isZIPMask(ShuffleMask, NumElts, WhichResult, OperandOrder);
240 if (!ZIPMask && !isZIP_v_undef_Mask(ShuffleMask, NumElts, WhichResult))
241 return false;
242 unsigned Opc = (WhichResult == 0) ? AArch64::G_ZIP1 : AArch64::G_ZIP2;
243 Register V1 = MI.getOperand(OperandOrder == 0 ? 1 : 2).getReg();
244 Register V2 = MI.getOperand(OperandOrder == 0 && ZIPMask ? 2 : 1).getReg();
245 MatchInfo = ShuffleVectorPseudo(Opc, Dst, {V1, V2});
246 return true;
247}
248
249/// Helper function for matchDup.
250bool matchDupFromInsertVectorElt(int Lane, MachineInstr &MI,
252 ShuffleVectorPseudo &MatchInfo) {
253 if (Lane != 0)
254 return false;
255
256 // Try to match a vector splat operation into a dup instruction.
257 // We're looking for this pattern:
258 //
259 // %scalar:gpr(s64) = COPY $x0
260 // %undef:fpr(<2 x s64>) = G_IMPLICIT_DEF
261 // %cst0:gpr(s32) = G_CONSTANT i32 0
262 // %zerovec:fpr(<2 x s32>) = G_BUILD_VECTOR %cst0(s32), %cst0(s32)
263 // %ins:fpr(<2 x s64>) = G_INSERT_VECTOR_ELT %undef, %scalar(s64), %cst0(s32)
264 // %splat:fpr(<2 x s64>) = G_SHUFFLE_VECTOR %ins(<2 x s64>), %undef,
265 // %zerovec(<2 x s32>)
266 //
267 // ...into:
268 // %splat = G_DUP %scalar
269
270 // Begin matching the insert.
271 auto *InsMI = getOpcodeDef(TargetOpcode::G_INSERT_VECTOR_ELT,
272 MI.getOperand(1).getReg(), MRI);
273 if (!InsMI)
274 return false;
275 // Match the undef vector operand.
276 if (!getOpcodeDef(TargetOpcode::G_IMPLICIT_DEF, InsMI->getOperand(1).getReg(),
277 MRI))
278 return false;
279
280 // Match the index constant 0.
281 if (!mi_match(InsMI->getOperand(3).getReg(), MRI, m_ZeroInt()))
282 return false;
283
284 MatchInfo = ShuffleVectorPseudo(AArch64::G_DUP, MI.getOperand(0).getReg(),
285 {InsMI->getOperand(2).getReg()});
286 return true;
287}
288
289/// Helper function for matchDup.
290bool matchDupFromBuildVector(int Lane, MachineInstr &MI,
292 ShuffleVectorPseudo &MatchInfo) {
293 assert(Lane >= 0 && "Expected positive lane?");
294 int NumElements = MRI.getType(MI.getOperand(1).getReg()).getNumElements();
295 // Test if the LHS is a BUILD_VECTOR. If it is, then we can just reference the
296 // lane's definition directly.
297 auto *BuildVecMI =
298 getOpcodeDef(TargetOpcode::G_BUILD_VECTOR,
299 MI.getOperand(Lane < NumElements ? 1 : 2).getReg(), MRI);
300 // If Lane >= NumElements then it is point to RHS, just check from RHS
301 if (NumElements <= Lane)
302 Lane -= NumElements;
303
304 if (!BuildVecMI)
305 return false;
306 Register Reg = BuildVecMI->getOperand(Lane + 1).getReg();
307 MatchInfo =
308 ShuffleVectorPseudo(AArch64::G_DUP, MI.getOperand(0).getReg(), {Reg});
309 return true;
310}
311
312bool matchDup(MachineInstr &MI, MachineRegisterInfo &MRI,
313 ShuffleVectorPseudo &MatchInfo) {
314 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
315 auto MaybeLane = getSplatIndex(MI);
316 if (!MaybeLane)
317 return false;
318 int Lane = *MaybeLane;
319 // If this is undef splat, generate it via "just" vdup, if possible.
320 if (Lane < 0)
321 Lane = 0;
322 if (matchDupFromInsertVectorElt(Lane, MI, MRI, MatchInfo))
323 return true;
324 if (matchDupFromBuildVector(Lane, MI, MRI, MatchInfo))
325 return true;
326 return false;
327}
328
329// Check if an EXT instruction can handle the shuffle mask when the vector
330// sources of the shuffle are the same.
331bool isSingletonExtMask(ArrayRef<int> M, LLT Ty) {
332 unsigned NumElts = Ty.getNumElements();
333
334 // Assume that the first shuffle index is not UNDEF. Fail if it is.
335 if (M[0] < 0)
336 return false;
337
338 // If this is a VEXT shuffle, the immediate value is the index of the first
339 // element. The other shuffle indices must be the successive elements after
340 // the first one.
341 unsigned ExpectedElt = M[0];
342 for (unsigned I = 1; I < NumElts; ++I) {
343 // Increment the expected index. If it wraps around, just follow it
344 // back to index zero and keep going.
345 ++ExpectedElt;
346 if (ExpectedElt == NumElts)
347 ExpectedElt = 0;
348
349 if (M[I] < 0)
350 continue; // Ignore UNDEF indices.
351 if (ExpectedElt != static_cast<unsigned>(M[I]))
352 return false;
353 }
354
355 return true;
356}
357
358bool matchEXT(MachineInstr &MI, MachineRegisterInfo &MRI,
359 ShuffleVectorPseudo &MatchInfo) {
360 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
361 Register Dst = MI.getOperand(0).getReg();
362 LLT DstTy = MRI.getType(Dst);
363 Register V1 = MI.getOperand(1).getReg();
364 Register V2 = MI.getOperand(2).getReg();
365 auto Mask = MI.getOperand(3).getShuffleMask();
367 auto ExtInfo = getExtMask(Mask, DstTy.getNumElements());
368 uint64_t ExtFactor = MRI.getType(V1).getScalarSizeInBits() / 8;
369
370 if (!ExtInfo) {
371 if (!getOpcodeDef<GImplicitDef>(V2, MRI) ||
372 !isSingletonExtMask(Mask, DstTy))
373 return false;
374
375 Imm = Mask[0] * ExtFactor;
376 MatchInfo = ShuffleVectorPseudo(AArch64::G_EXT, Dst, {V1, V1, Imm});
377 return true;
378 }
379 bool ReverseExt;
380 std::tie(ReverseExt, Imm) = *ExtInfo;
381 if (ReverseExt)
382 std::swap(V1, V2);
383 Imm *= ExtFactor;
384 MatchInfo = ShuffleVectorPseudo(AArch64::G_EXT, Dst, {V1, V2, Imm});
385 return true;
386}
387
388/// Replace a G_SHUFFLE_VECTOR instruction with a pseudo.
389/// \p Opc is the opcode to use. \p MI is the G_SHUFFLE_VECTOR.
390void applyShuffleVectorPseudo(MachineInstr &MI, MachineRegisterInfo &MRI,
391 ShuffleVectorPseudo &MatchInfo) {
392 MachineIRBuilder MIRBuilder(MI);
393 if (MatchInfo.Opc == TargetOpcode::G_BSWAP) {
394 assert(MatchInfo.SrcOps.size() == 1);
395 LLT DstTy = MRI.getType(MatchInfo.Dst);
396 assert(DstTy == LLT::fixed_vector(8, 8) ||
397 DstTy == LLT::fixed_vector(16, 8));
398 LLT BSTy = DstTy == LLT::fixed_vector(8, 8)
401 // FIXME: NVCAST
402 auto BS1 = MIRBuilder.buildInstr(TargetOpcode::G_BITCAST, {BSTy},
403 MatchInfo.SrcOps[0]);
404 auto BS2 = MIRBuilder.buildInstr(MatchInfo.Opc, {BSTy}, {BS1});
405 MIRBuilder.buildInstr(TargetOpcode::G_BITCAST, {MatchInfo.Dst}, {BS2});
406 } else
407 MIRBuilder.buildInstr(MatchInfo.Opc, {MatchInfo.Dst}, MatchInfo.SrcOps);
408 MI.eraseFromParent();
409}
410
411/// Replace a G_SHUFFLE_VECTOR instruction with G_EXT.
412/// Special-cased because the constant operand must be emitted as a G_CONSTANT
413/// for the imported tablegen patterns to work.
414void applyEXT(MachineInstr &MI, ShuffleVectorPseudo &MatchInfo) {
415 MachineIRBuilder MIRBuilder(MI);
416 if (MatchInfo.SrcOps[2].getImm() == 0)
417 MIRBuilder.buildCopy(MatchInfo.Dst, MatchInfo.SrcOps[0]);
418 else {
419 // Tablegen patterns expect an i32 G_CONSTANT as the final op.
420 auto Cst = MIRBuilder.buildConstant(LLT::integer(32),
421 MatchInfo.SrcOps[2].getImm());
422 MIRBuilder.buildInstr(MatchInfo.Opc, {MatchInfo.Dst},
423 {MatchInfo.SrcOps[0], MatchInfo.SrcOps[1], Cst});
424 }
425 MI.eraseFromParent();
426}
427
428void applyFullRev(MachineInstr &MI, MachineRegisterInfo &MRI) {
429 Register Dst = MI.getOperand(0).getReg();
430 Register Src = MI.getOperand(1).getReg();
431 LLT DstTy = MRI.getType(Dst);
432 assert(DstTy.getSizeInBits() == 128 &&
433 "Expected 128bit vector in applyFullRev");
434 MachineIRBuilder MIRBuilder(MI);
435 auto Cst = MIRBuilder.buildConstant(LLT::integer(32), 8);
436 auto Rev = MIRBuilder.buildInstr(AArch64::G_REV64, {DstTy}, {Src});
437 MIRBuilder.buildInstr(AArch64::G_EXT, {Dst}, {Rev, Rev, Cst});
438 MI.eraseFromParent();
439}
440
441bool matchNonConstInsert(MachineInstr &MI, MachineRegisterInfo &MRI) {
442 assert(MI.getOpcode() == TargetOpcode::G_INSERT_VECTOR_ELT);
443
444 auto ValAndVReg =
445 getIConstantVRegValWithLookThrough(MI.getOperand(3).getReg(), MRI);
446 return !ValAndVReg;
447}
448
449void applyNonConstInsert(MachineInstr &MI, MachineRegisterInfo &MRI,
450 MachineIRBuilder &Builder) {
451 auto &Insert = cast<GInsertVectorElement>(MI);
452 Builder.setInstrAndDebugLoc(Insert);
453
454 Register Offset = Insert.getIndexReg();
455 LLT VecTy = MRI.getType(Insert.getReg(0));
456 LLT EltTy = MRI.getType(Insert.getElementReg());
457 LLT IdxTy = MRI.getType(Insert.getIndexReg());
458
459 if (VecTy.isScalableVector())
460 return;
461
462 // Create a stack slot and store the vector into it
463 MachineFunction &MF = Builder.getMF();
464 Align Alignment(
465 std::min<uint64_t>(VecTy.getSizeInBytes().getKnownMinValue(), 16));
466 int FrameIdx = MF.getFrameInfo().CreateStackObject(VecTy.getSizeInBytes(),
467 Alignment, false);
468 LLT FramePtrTy = LLT::pointer(0, 64);
470 auto StackTemp = Builder.buildFrameIndex(FramePtrTy, FrameIdx);
471
472 Builder.buildStore(Insert.getOperand(1), StackTemp, PtrInfo, Align(8));
473
474 // Get the pointer to the element, and be sure not to hit undefined behavior
475 // if the index is out of bounds.
477 "Expected a power-2 vector size");
478 auto Mask = Builder.buildConstant(IdxTy, VecTy.getNumElements() - 1);
479 Register And = Builder.buildAnd(IdxTy, Offset, Mask).getReg(0);
480 auto EltSize = Builder.buildConstant(IdxTy, EltTy.getSizeInBytes());
481 Register Mul = Builder.buildMul(IdxTy, And, EltSize).getReg(0);
482 Register EltPtr =
483 Builder.buildPtrAdd(MRI.getType(StackTemp.getReg(0)), StackTemp, Mul)
484 .getReg(0);
485
486 // Write the inserted element
487 Builder.buildStore(Insert.getElementReg(), EltPtr, PtrInfo, Align(1));
488 // Reload the whole vector.
489 Builder.buildLoad(Insert.getReg(0), StackTemp, PtrInfo, Align(8));
490 Insert.eraseFromParent();
491}
492
493/// Match a G_SHUFFLE_VECTOR with a mask which corresponds to a
494/// G_INSERT_VECTOR_ELT and G_EXTRACT_VECTOR_ELT pair.
495///
496/// e.g.
497/// %shuf = G_SHUFFLE_VECTOR %left, %right, shufflemask(0, 0)
498///
499/// Can be represented as
500///
501/// %extract = G_EXTRACT_VECTOR_ELT %left, 0
502/// %ins = G_INSERT_VECTOR_ELT %left, %extract, 1
503///
504bool matchINS(MachineInstr &MI, MachineRegisterInfo &MRI,
505 std::tuple<Register, int, Register, int> &MatchInfo) {
506 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
507 ArrayRef<int> ShuffleMask = MI.getOperand(3).getShuffleMask();
508 Register Dst = MI.getOperand(0).getReg();
509 int NumElts = MRI.getType(Dst).getNumElements();
510 auto DstIsLeftAndDstLane = isINSMask(ShuffleMask, NumElts);
511 if (!DstIsLeftAndDstLane)
512 return false;
513 bool DstIsLeft;
514 int DstLane;
515 std::tie(DstIsLeft, DstLane) = *DstIsLeftAndDstLane;
516 Register Left = MI.getOperand(1).getReg();
517 Register Right = MI.getOperand(2).getReg();
518 Register DstVec = DstIsLeft ? Left : Right;
519 Register SrcVec = Left;
520
521 int SrcLane = ShuffleMask[DstLane];
522 if (SrcLane >= NumElts) {
523 SrcVec = Right;
524 SrcLane -= NumElts;
525 }
526
527 MatchInfo = std::make_tuple(DstVec, DstLane, SrcVec, SrcLane);
528 return true;
529}
530
531void applyINS(MachineInstr &MI, MachineRegisterInfo &MRI,
532 MachineIRBuilder &Builder,
533 std::tuple<Register, int, Register, int> &MatchInfo) {
534 Builder.setInstrAndDebugLoc(MI);
535 Register Dst = MI.getOperand(0).getReg();
536 auto ScalarTy = MRI.getType(Dst).getElementType();
537 Register DstVec, SrcVec;
538 int DstLane, SrcLane;
539 std::tie(DstVec, DstLane, SrcVec, SrcLane) = MatchInfo;
540 auto SrcCst = Builder.buildConstant(LLT::integer(64), SrcLane);
541 auto Extract = Builder.buildExtractVectorElement(ScalarTy, SrcVec, SrcCst);
542 auto DstCst = Builder.buildConstant(LLT::integer(64), DstLane);
543 Builder.buildInsertVectorElement(Dst, DstVec, Extract, DstCst);
544 MI.eraseFromParent();
545}
546
547/// isVShiftRImm - Check if this is a valid vector for the immediate
548/// operand of a vector shift right operation. The value must be in the range:
549/// 1 <= Value <= ElementBits for a right shift.
551 int64_t &Cnt) {
552 assert(Ty.isVector() && "vector shift count is not a vector type");
554 auto Cst = getAArch64VectorSplatScalar(*MI, MRI);
555 if (!Cst)
556 return false;
557 Cnt = *Cst;
558 int64_t ElementBits = Ty.getScalarSizeInBits();
559 return Cnt >= 1 && Cnt <= ElementBits;
560}
561
562/// Match a vector G_ASHR or G_LSHR with a valid immediate shift.
563bool matchVAshrLshrImm(MachineInstr &MI, MachineRegisterInfo &MRI,
564 int64_t &Imm) {
565 assert(MI.getOpcode() == TargetOpcode::G_ASHR ||
566 MI.getOpcode() == TargetOpcode::G_LSHR);
567 LLT Ty = MRI.getType(MI.getOperand(1).getReg());
568 if (!Ty.isVector())
569 return false;
570 return isVShiftRImm(MI.getOperand(2).getReg(), MRI, Ty, Imm);
571}
572
573void applyVAshrLshrImm(MachineInstr &MI, MachineRegisterInfo &MRI,
574 int64_t &Imm) {
575 unsigned Opc = MI.getOpcode();
576 assert(Opc == TargetOpcode::G_ASHR || Opc == TargetOpcode::G_LSHR);
577 unsigned NewOpc =
578 Opc == TargetOpcode::G_ASHR ? AArch64::G_VASHR : AArch64::G_VLSHR;
579 MachineIRBuilder MIB(MI);
580 MIB.buildInstr(NewOpc, {MI.getOperand(0)}, {MI.getOperand(1)}).addImm(Imm);
581 MI.eraseFromParent();
582}
583
584/// Determine whether an integer G_ICMP against 1 or -1 can compare
585/// against 0 instead.
586///
587/// AArch64 can fold a compare-with-zero more cheaply than some non-arithmetic
588/// immediates (SUBS/ADDS, or TST when the LHS is an AND). When the predicate
589/// can be adjusted without changing semantics, the RHS may become 0.
590///
591/// Supported transforms (signed predicates only):
592/// (and X, Y) slt 1 => (and X, Y) sle 0
593/// (and X, Y) sge 1 => (and X, Y) sgt 0
594/// X sle -1 => X slt 0
595/// X sgt -1 => X sge 0
596///
597/// The compare-against-1 cases require the LHS to be G_AND because the
598/// compare-with-zero path enables ANDS (TST) selection, and ANDS flags are
599/// only reliable for those signed comparisons. This mirrors SelectionDAG
600/// emitComparison().
601///
602/// For compare-against--1 on a non-AND LHS, \p LHS must have a single
603/// non-debug use so other users are not left with a different immediate.
604///
605/// \param LHS The compare LHS register.
606/// \param C The constant RHS (only 1 or all-ones are considered).
607/// \param P In/out predicate; updated when a transform applies.
608/// \param MRI Used to inspect the LHS definition and use count.
609/// \returns true if \p P was updated and comparing against 0 is equivalent.
610static bool shouldBeAdjustedToZero(Register LHS, const APInt &C,
612 const MachineRegisterInfo &MRI) {
613 const bool IsAndLHS = getOpcodeDef<GAnd>(LHS, MRI) != nullptr;
614
615 if (C.isOne() && (P == CmpInst::ICMP_SLT || P == CmpInst::ICMP_SGE) &&
616 IsAndLHS) {
618 return true;
619 }
620
621 if (!IsAndLHS && !MRI.hasOneNonDBGUse(LHS))
622 return false;
623
624 if (C.isAllOnes() && (P == CmpInst::ICMP_SLE || P == CmpInst::ICMP_SGT)) {
626 return true;
627 }
628 return false;
629}
630
631/// Determine if it is possible to modify the \p RHS and predicate \p P of a
632/// G_ICMP instruction such that the right-hand side is an arithmetic immediate.
633///
634/// \returns A pair containing the updated immediate and predicate which may
635/// be used to optimize the instruction.
636///
637/// \note This assumes that the comparison has been legalized.
638std::optional<std::pair<uint64_t, CmpInst::Predicate>>
639tryAdjustICmpImmAndPred(Register LHS, Register RHS, CmpInst::Predicate P,
640 const MachineRegisterInfo &MRI) {
641 const auto &Ty = MRI.getType(RHS);
642 if (Ty.isVector())
643 return std::nullopt;
644 assert((Ty.getSizeInBits() == 32 || Ty.getSizeInBits() == 64) &&
645 "Expected 32 or 64 bit compare only?");
646
647 // If the RHS is not a constant, or the RHS is already a valid arithmetic
648 // immediate, then there is nothing to change.
649 auto ValAndVReg = getIConstantVRegValWithLookThrough(RHS, MRI);
650 if (!ValAndVReg)
651 return std::nullopt;
652 APInt C = ValAndVReg->Value;
653 if (shouldBeAdjustedToZero(LHS, C, P, MRI))
654 return {{0, P}};
655
657 return std::nullopt;
658
659 uint64_t OriginalC = C.getZExtValue();
660
661 // We have a non-arithmetic immediate. Check if adjusting the immediate and
662 // adjusting the predicate will result in a legal arithmetic immediate.
663 switch (P) {
664 default:
665 return std::nullopt;
668 // Check for
669 //
670 // x slt c => x sle c - 1
671 // x sge c => x sgt c - 1
672 //
673 // When c is not the smallest possible negative number.
674 if (C.isMinSignedValue())
675 return std::nullopt;
677 C = C - 1;
678 break;
681 // Check for
682 //
683 // x ult c => x ule c - 1
684 // x uge c => x ugt c - 1
685 //
686 // When c is not zero.
687 assert(!C.isZero() && "C should not be zero here!");
689 C = C - 1;
690 break;
693 // Check for
694 //
695 // x sle c => x slt c + 1
696 // x sgt c => s sge c + 1
697 //
698 // When c is not the largest possible signed integer.
699 if (C.isMaxSignedValue())
700 return std::nullopt;
702 C = C + 1;
703 break;
706 // Check for
707 //
708 // x ule c => x ult c + 1
709 // x ugt c => s uge c + 1
710 //
711 // When c is not the largest possible unsigned integer.
712 if (C.isAllOnes())
713 return std::nullopt;
715 C = C + 1;
716 break;
717 }
718
719 // Check if the new constant is valid, and return the updated constant and
720 // predicate if it is.
721 uint64_t NewC = C.getZExtValue();
723 return {{NewC, P}};
724
725 auto NumberOfInstrToLoadImm = [=](uint64_t Imm) {
728 return Insn.size();
729 };
730
731 if (NumberOfInstrToLoadImm(OriginalC) > NumberOfInstrToLoadImm(NewC))
732 return {{NewC, P}};
733
734 return std::nullopt;
735}
736
737/// Determine whether or not it is possible to update the RHS and predicate of
738/// a G_ICMP instruction such that the RHS will be selected as an arithmetic
739/// immediate.
740///
741/// \p MI - The G_ICMP instruction
742/// \p MatchInfo - The new RHS immediate and predicate on success
743///
744/// See tryAdjustICmpImmAndPred for valid transformations.
745bool matchAdjustICmpImmAndPred(
747 std::pair<uint64_t, CmpInst::Predicate> &MatchInfo) {
748 assert(MI.getOpcode() == TargetOpcode::G_ICMP);
749 Register LHS = MI.getOperand(2).getReg();
750 Register RHS = MI.getOperand(3).getReg();
751 auto Pred = static_cast<CmpInst::Predicate>(MI.getOperand(1).getPredicate());
752 if (auto MaybeNewImmAndPred = tryAdjustICmpImmAndPred(LHS, RHS, Pred, MRI)) {
753 MatchInfo = *MaybeNewImmAndPred;
754 return true;
755 }
756 return false;
757}
758
759void applyAdjustICmpImmAndPred(
760 MachineInstr &MI, std::pair<uint64_t, CmpInst::Predicate> &MatchInfo,
761 MachineIRBuilder &MIB, GISelChangeObserver &Observer) {
763 MachineOperand &RHS = MI.getOperand(3);
764 MachineRegisterInfo &MRI = *MIB.getMRI();
765 auto Cst = MIB.buildConstant(MRI.cloneVirtualRegister(RHS.getReg()),
766 MatchInfo.first);
767 Observer.changingInstr(MI);
768 RHS.setReg(Cst->getOperand(0).getReg());
769 MI.getOperand(1).setPredicate(MatchInfo.second);
770 Observer.changedInstr(MI);
771}
772
773bool matchDupLane(MachineInstr &MI, MachineRegisterInfo &MRI,
774 std::pair<unsigned, int> &MatchInfo) {
775 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
776 Register Src1Reg = MI.getOperand(1).getReg();
777 const LLT SrcTy = MRI.getType(Src1Reg);
778 const LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
779
780 auto LaneIdx = getSplatIndex(MI);
781 if (!LaneIdx)
782 return false;
783
784 // The lane idx should be within the first source vector.
785 if (*LaneIdx >= SrcTy.getNumElements())
786 return false;
787
788 if (DstTy != SrcTy)
789 return false;
790
791 LLT ScalarTy = SrcTy.getElementType();
792 unsigned ScalarSize = ScalarTy.getSizeInBits();
793
794 unsigned Opc = 0;
795 switch (SrcTy.getNumElements()) {
796 case 2:
797 if (ScalarSize == 64)
798 Opc = AArch64::G_DUPLANE64;
799 else if (ScalarSize == 32)
800 Opc = AArch64::G_DUPLANE32;
801 break;
802 case 4:
803 if (ScalarSize == 32)
804 Opc = AArch64::G_DUPLANE32;
805 else if (ScalarSize == 16)
806 Opc = AArch64::G_DUPLANE16;
807 break;
808 case 8:
809 if (ScalarSize == 8)
810 Opc = AArch64::G_DUPLANE8;
811 else if (ScalarSize == 16)
812 Opc = AArch64::G_DUPLANE16;
813 break;
814 case 16:
815 if (ScalarSize == 8)
816 Opc = AArch64::G_DUPLANE8;
817 break;
818 default:
819 break;
820 }
821 if (!Opc)
822 return false;
823
824 MatchInfo.first = Opc;
825 MatchInfo.second = *LaneIdx;
826 return true;
827}
828
829void applyDupLane(MachineInstr &MI, MachineRegisterInfo &MRI,
830 MachineIRBuilder &B, std::pair<unsigned, int> &MatchInfo) {
831 assert(MI.getOpcode() == TargetOpcode::G_SHUFFLE_VECTOR);
832 Register Src1Reg = MI.getOperand(1).getReg();
833 const LLT SrcTy = MRI.getType(Src1Reg);
834
835 B.setInstrAndDebugLoc(MI);
836 auto Lane = B.buildConstant(LLT::integer(64), MatchInfo.second);
837
838 Register DupSrc = MI.getOperand(1).getReg();
839 // For types like <2 x s32>, we can use G_DUPLANE32, with a <4 x s32> source.
840 // To do this, we can use a G_CONCAT_VECTORS to do the widening.
841 if (SrcTy.getSizeInBits() == 64) {
842 auto Undef = B.buildUndef(SrcTy);
843 DupSrc = B.buildConcatVectors(SrcTy.multiplyElements(2),
844 {Src1Reg, Undef.getReg(0)})
845 .getReg(0);
846 }
847 B.buildInstr(MatchInfo.first, {MI.getOperand(0).getReg()}, {DupSrc, Lane});
848 MI.eraseFromParent();
849}
850
851bool matchScalarizeVectorUnmerge(MachineInstr &MI, MachineRegisterInfo &MRI) {
852 auto &Unmerge = cast<GUnmerge>(MI);
853 Register Src1Reg = Unmerge.getReg(Unmerge.getNumOperands() - 1);
854 const LLT SrcTy = MRI.getType(Src1Reg);
855 if (SrcTy.getSizeInBits() != 128 && SrcTy.getSizeInBits() != 64)
856 return false;
857 return SrcTy.isVector() && !SrcTy.isScalable() &&
858 (Unmerge.getNumOperands() == (unsigned)SrcTy.getNumElements() + 1 ||
859 (Unmerge.getNumDefs() == 2 && SrcTy.getSizeInBits() == 128 &&
860 MRI.getType(Unmerge.getReg(0)).getSizeInBits() == 64));
861}
862
863void applyScalarizeVectorUnmerge(MachineInstr &MI, MachineRegisterInfo &MRI,
865 auto &Unmerge = cast<GUnmerge>(MI);
866 Register Src1Reg = Unmerge.getReg(Unmerge.getNumOperands() - 1);
867 const LLT SrcTy = MRI.getType(Src1Reg);
868 const LLT DstTy = MRI.getType(Unmerge.getReg(0));
869 assert((SrcTy.isVector() && !SrcTy.isScalable()) &&
870 "Expected a fixed length vector");
871
872 if (DstTy.isVector()) {
873 assert(Unmerge.getNumDefs() == 2);
874 if (!MRI.use_nodbg_empty(Unmerge.getReg(0)))
875 B.buildExtractSubvector(Unmerge.getReg(0), Src1Reg, 0);
876 if (!MRI.use_nodbg_empty(Unmerge.getReg(1)))
877 B.buildExtractSubvector(Unmerge.getReg(1), Src1Reg,
878 SrcTy.getNumElements() / 2);
879 } else {
880 for (int I = 0; I < SrcTy.getNumElements(); ++I)
881 if (!MRI.use_nodbg_empty(Unmerge.getReg(I)))
882 B.buildExtractVectorElementConstant(Unmerge.getReg(I), Src1Reg, I);
883 }
884 MI.eraseFromParent();
885}
886
887bool matchBuildVectorToDup(MachineInstr &MI, Register &Src,
888 MachineRegisterInfo &MRI) {
889 assert(MI.getOpcode() == TargetOpcode::G_BUILD_VECTOR);
890
891 // Later, during selection, we'll try to match imported patterns using
892 // immAllOnesV and immAllZerosV. These require G_BUILD_VECTOR. Don't lower
893 // G_BUILD_VECTORs which could match those patterns.
895 return false;
896
897 // Find buildvector which always uses the same register or undef. Return true
898 // so long as at least 2 registers were found (not all-undef or only 1
899 // non-undef entry).
900 Register Reg = 0;
901 unsigned NumNonUndef = 0;
902 for (const MachineOperand &Op : drop_begin(MI.operands())) {
903 if (getOpcodeDef<GImplicitDef>(Op.getReg(), MRI))
904 continue;
905
906 if (!Reg)
907 Reg = Op.getReg();
908 else if (Op.getReg() != Reg)
909 return false;
910 NumNonUndef++;
911 }
912
913 Src = Reg;
914 return Reg && NumNonUndef > 1;
915}
916
917void applyBuildVectorToDup(MachineInstr &MI, Register Src,
919 B.setInstrAndDebugLoc(MI);
920 B.buildInstr(AArch64::G_DUP, {MI.getOperand(0).getReg()}, {Src});
921 MI.eraseFromParent();
922}
923
924/// \returns how many instructions would be saved by folding a G_ICMP's shift
925/// and/or extension operations.
926static unsigned getCmpOperandFoldingProfit(Register CmpOp,
927 MachineRegisterInfo &MRI) {
928 // FIXME: This is duplicated with the selector. (See: selectShiftedRegister)
929 auto IsSupportedExtend = [&](const MachineInstr &MI) {
930 if (MI.getOpcode() == TargetOpcode::G_SEXT_INREG)
931 return true;
932 if (MI.getOpcode() == TargetOpcode::G_AND) {
933 auto ValAndVReg =
934 getIConstantVRegValWithLookThrough(MI.getOperand(2).getReg(), MRI);
935 if (ValAndVReg) {
936 uint64_t Mask = ValAndVReg->Value.getZExtValue();
937 return (Mask == 0xFF || Mask == 0xFFFF || Mask == 0xFFFFFFFF);
938 }
939 }
940 return false;
941 };
942
943 // No instructions to save if there's more than one use or no uses.
944 if (!MRI.hasOneNonDBGUse(CmpOp))
945 return 0;
946
947 MachineInstr *Def = getDefIgnoringCopies(CmpOp, MRI);
948 if (IsSupportedExtend(*Def))
949 return 1;
950
951 unsigned Opc = Def->getOpcode();
952 if (Opc == TargetOpcode::G_SHL || Opc == TargetOpcode::G_LSHR ||
953 Opc == TargetOpcode::G_ASHR) {
954 auto MaybeShiftAmt =
955 getIConstantVRegValWithLookThrough(Def->getOperand(2).getReg(), MRI);
956 if (MaybeShiftAmt) {
957 uint64_t ShiftAmt = MaybeShiftAmt->Value.getZExtValue();
958 MachineInstr *ShiftLHS =
959 getDefIgnoringCopies(Def->getOperand(1).getReg(), MRI);
960 if (IsSupportedExtend(*ShiftLHS))
961 return (ShiftAmt <= 4) ? 2 : 1;
962 LLT Ty = MRI.getType(Def->getOperand(0).getReg());
963 if (Ty.isVector())
964 return 0;
965 unsigned ShiftSize = Ty.getSizeInBits();
966 if ((ShiftSize == 32 && ShiftAmt <= 31) ||
967 (ShiftSize == 64 && ShiftAmt <= 63))
968 return 1;
969 }
970 }
971
972 return 0;
973}
974
975/// \returns true if it would be profitable to swap the LHS and RHS of a G_ICMP
976/// instruction \p MI.
977bool trySwapICmpOperands(MachineInstr &MI, MachineRegisterInfo &MRI) {
978 assert(MI.getOpcode() == TargetOpcode::G_ICMP);
979 // Swap the operands if it would introduce a profitable folding opportunity.
980 // (e.g. a shift + extend).
981 //
982 // For example:
983 // lsl w13, w11, #1
984 // cmp w13, w12
985 // can be turned into:
986 // cmp w12, w11, lsl #1
987
988 // Don't swap if there's a constant on the RHS and it is a legal compare
989 // immediate, because we know we can fold that.
990 Register RHS = MI.getOperand(3).getReg();
991 auto RHSCst = getIConstantVRegValWithLookThrough(RHS, MRI);
992 if (RHSCst && AArch64_AM::isLegalCmpImmed(RHSCst->Value))
993 return false;
994
995 Register LHS = MI.getOperand(2).getReg();
996 auto Pred = static_cast<CmpInst::Predicate>(MI.getOperand(1).getPredicate());
997 auto GetRegForProfit = [&](Register Reg) {
999 return isCMN(Def, Pred, MRI) ? Def->getOperand(2).getReg() : Reg;
1000 };
1001
1002 // Don't have a constant on the RHS. If we swap the LHS and RHS of the
1003 // compare, would we be able to fold more instructions?
1004 Register TheLHS = GetRegForProfit(LHS);
1005 Register TheRHS = GetRegForProfit(RHS);
1006
1007 // If the LHS is more likely to give us a folding opportunity, then swap the
1008 // LHS and RHS.
1009 return (getCmpOperandFoldingProfit(TheLHS, MRI) >
1010 getCmpOperandFoldingProfit(TheRHS, MRI));
1011}
1012
1013void applySwapICmpOperands(MachineInstr &MI, GISelChangeObserver &Observer) {
1014 auto Pred = static_cast<CmpInst::Predicate>(MI.getOperand(1).getPredicate());
1015 Register LHS = MI.getOperand(2).getReg();
1016 Register RHS = MI.getOperand(3).getReg();
1017 Observer.changingInstr(MI);
1018 MI.getOperand(1).setPredicate(CmpInst::getSwappedPredicate(Pred));
1019 MI.getOperand(2).setReg(RHS);
1020 MI.getOperand(3).setReg(LHS);
1021 Observer.changedInstr(MI);
1022}
1023
1024/// \returns a function which builds a vector floating point compare instruction
1025/// for a condition code \p CC.
1026/// \param [in] NoNans - True if the instruction has nnan flag.
1027std::function<Register(MachineIRBuilder &)>
1028getVectorFCMP(AArch64CC::CondCode CC, Register LHS, Register RHS, bool NoNans,
1029 MachineRegisterInfo &MRI) {
1030 LLT OldTy = MRI.getType(LHS);
1031 LLT DstTy = LLT::fixed_vector(OldTy.getNumElements(),
1033 assert(DstTy.isVector() && "Expected vector types only?");
1034 switch (CC) {
1035 default:
1036 llvm_unreachable("Unexpected condition code!");
1037 case AArch64CC::NE:
1038 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1039 auto FCmp = MIB.buildInstr(AArch64::G_FCMEQ, {DstTy}, {LHS, RHS});
1040 return MIB.buildNot(DstTy, FCmp).getReg(0);
1041 };
1042 case AArch64CC::EQ:
1043 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1044 return MIB.buildInstr(AArch64::G_FCMEQ, {DstTy}, {LHS, RHS}).getReg(0);
1045 };
1046 case AArch64CC::GE:
1047 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1048 return MIB.buildInstr(AArch64::G_FCMGE, {DstTy}, {LHS, RHS}).getReg(0);
1049 };
1050 case AArch64CC::GT:
1051 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1052 return MIB.buildInstr(AArch64::G_FCMGT, {DstTy}, {LHS, RHS}).getReg(0);
1053 };
1054 case AArch64CC::LS:
1055 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1056 return MIB.buildInstr(AArch64::G_FCMGE, {DstTy}, {RHS, LHS}).getReg(0);
1057 };
1058 case AArch64CC::MI:
1059 return [LHS, RHS, DstTy](MachineIRBuilder &MIB) {
1060 return MIB.buildInstr(AArch64::G_FCMGT, {DstTy}, {RHS, LHS}).getReg(0);
1061 };
1062 }
1063}
1064
1065/// Try to lower a vector G_FCMP \p MI into an AArch64-specific pseudo.
1066bool matchLowerVectorFCMP(MachineInstr &MI, MachineRegisterInfo &MRI,
1067 MachineIRBuilder &MIB) {
1068 assert(MI.getOpcode() == TargetOpcode::G_FCMP);
1069 const auto &ST = MI.getMF()->getSubtarget<AArch64Subtarget>();
1070
1071 Register Dst = MI.getOperand(0).getReg();
1072 LLT DstTy = MRI.getType(Dst);
1073 if (!DstTy.isVector() || !ST.hasNEON())
1074 return false;
1075 Register LHS = MI.getOperand(2).getReg();
1076 unsigned EltSize = MRI.getType(LHS).getScalarSizeInBits();
1077 if (EltSize == 16 && !ST.hasFullFP16())
1078 return false;
1079 if (EltSize != 16 && EltSize != 32 && EltSize != 64)
1080 return false;
1081
1082 return true;
1083}
1084
1085/// Try to lower a vector G_FCMP \p MI into an AArch64-specific pseudo.
1086void applyLowerVectorFCMP(MachineInstr &MI, MachineRegisterInfo &MRI,
1087 MachineIRBuilder &MIB) {
1088 assert(MI.getOpcode() == TargetOpcode::G_FCMP);
1089
1090 const auto &CmpMI = cast<GFCmp>(MI);
1091
1092 Register Dst = CmpMI.getReg(0);
1093 CmpInst::Predicate Pred = CmpMI.getCond();
1094 Register LHS = CmpMI.getLHSReg();
1095 Register RHS = CmpMI.getRHSReg();
1096
1097 LLT DstTy = MRI.getType(Dst);
1098
1099 bool Invert = false;
1101 if ((Pred == CmpInst::Predicate::FCMP_ORD ||
1103 isBuildVectorAllZeros(*MRI.getVRegDef(RHS), MRI)) {
1104 // The special case "fcmp ord %a, 0" is the canonical check that LHS isn't
1105 // NaN, so equivalent to a == a and doesn't need the two comparisons an
1106 // "ord" normally would.
1107 // Similarly, "fcmp uno %a, 0" is the canonical check that LHS is NaN and is
1108 // thus equivalent to a != a.
1109 RHS = LHS;
1111 } else
1112 changeVectorFCMPPredToAArch64CC(Pred, CC, CC2, Invert);
1113
1114 // Instead of having an apply function, just build here to simplify things.
1116
1117 // TODO: Also consider GISelValueTracking result if eligible.
1118 const bool NoNans = MI.getFlag(MachineInstr::FmNoNans);
1119
1120 auto Cmp = getVectorFCMP(CC, LHS, RHS, NoNans, MRI);
1121 Register CmpRes;
1122 if (CC2 == AArch64CC::AL)
1123 CmpRes = Cmp(MIB);
1124 else {
1125 auto Cmp2 = getVectorFCMP(CC2, LHS, RHS, NoNans, MRI);
1126 auto Cmp2Dst = Cmp2(MIB);
1127 auto Cmp1Dst = Cmp(MIB);
1128 CmpRes = MIB.buildOr(DstTy, Cmp1Dst, Cmp2Dst).getReg(0);
1129 }
1130 if (Invert)
1131 CmpRes = MIB.buildNot(DstTy, CmpRes).getReg(0);
1132 MRI.replaceRegWith(Dst, CmpRes);
1133 MI.eraseFromParent();
1134}
1135
1136// Matches G_BUILD_VECTOR where at least one source operand is not a constant
1137bool matchLowerBuildToInsertVecElt(MachineInstr &MI, MachineRegisterInfo &MRI) {
1138 auto *GBuildVec = cast<GBuildVector>(&MI);
1139
1140 // Check if the values are all constants
1141 for (unsigned I = 0; I < GBuildVec->getNumSources(); ++I) {
1142 auto ConstVal =
1143 getAnyConstantVRegValWithLookThrough(GBuildVec->getSourceReg(I), MRI);
1144
1145 if (!ConstVal.has_value())
1146 return true;
1147 }
1148
1149 return false;
1150}
1151
1152void applyLowerBuildToInsertVecElt(MachineInstr &MI, MachineRegisterInfo &MRI,
1154 auto *GBuildVec = cast<GBuildVector>(&MI);
1155 LLT DstTy = MRI.getType(GBuildVec->getReg(0));
1156 Register DstReg = B.buildUndef(DstTy).getReg(0);
1157
1158 for (unsigned I = 0; I < GBuildVec->getNumSources(); ++I) {
1159 Register SrcReg = GBuildVec->getSourceReg(I);
1160 if (mi_match(SrcReg, MRI, m_GImplicitDef()))
1161 continue;
1162 auto IdxReg = B.buildConstant(LLT::integer(64), I);
1163 DstReg =
1164 B.buildInsertVectorElement(DstTy, DstReg, SrcReg, IdxReg).getReg(0);
1165 }
1166 B.buildCopy(GBuildVec->getReg(0), DstReg);
1167 GBuildVec->eraseFromParent();
1168}
1169
1170bool matchFormTruncstore(MachineInstr &MI, MachineRegisterInfo &MRI,
1171 Register &SrcReg) {
1172 assert(MI.getOpcode() == TargetOpcode::G_STORE);
1173 Register DstReg = MI.getOperand(0).getReg();
1174 if (cast<GLoadStore>(MI).isAtomic())
1175 return false;
1176 if (MRI.getType(DstReg).isVector())
1177 return false;
1178 // Match a store of a truncate.
1179 if (!mi_match(DstReg, MRI, m_GTrunc(m_Reg(SrcReg))))
1180 return false;
1181 // Only form truncstores for value types of max 64b.
1182 return MRI.getType(SrcReg).getSizeInBits() <= 64;
1183}
1184
1185void applyFormTruncstore(MachineInstr &MI, MachineRegisterInfo &MRI,
1187 Register &SrcReg) {
1188 assert(MI.getOpcode() == TargetOpcode::G_STORE);
1189 Observer.changingInstr(MI);
1190 MI.getOperand(0).setReg(SrcReg);
1191 Observer.changedInstr(MI);
1192}
1193
1194// Lower vector G_SEXT_INREG back to shifts for selection. We allowed them to
1195// form in the first place for combine opportunities, so any remaining ones
1196// at this stage need be lowered back.
1197bool matchVectorSextInReg(MachineInstr &MI, MachineRegisterInfo &MRI) {
1198 assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
1199 Register DstReg = MI.getOperand(0).getReg();
1200 LLT DstTy = MRI.getType(DstReg);
1201 return DstTy.isVector();
1202}
1203
1204void applyVectorSextInReg(MachineInstr &MI, MachineRegisterInfo &MRI,
1206 assert(MI.getOpcode() == TargetOpcode::G_SEXT_INREG);
1207 B.setInstrAndDebugLoc(MI);
1208 LegalizerHelper Helper(*MI.getMF(), Observer, B);
1209 Helper.lower(MI, 0, /* Unused hint type */ LLT());
1210}
1211
1212/// Combine <N x t>, unused = unmerge(G_EXT <2*N x t> v, undef, N)
1213/// => unused, <N x t> = unmerge v
1214bool matchUnmergeExtToUnmerge(MachineInstr &MI, MachineRegisterInfo &MRI,
1215 Register &MatchInfo) {
1216 auto &Unmerge = cast<GUnmerge>(MI);
1217 if (Unmerge.getNumDefs() != 2)
1218 return false;
1219 if (!MRI.use_nodbg_empty(Unmerge.getReg(1)))
1220 return false;
1221
1222 LLT DstTy = MRI.getType(Unmerge.getReg(0));
1223 if (!DstTy.isVector())
1224 return false;
1225
1226 MachineInstr *Ext = getOpcodeDef(AArch64::G_EXT, Unmerge.getSourceReg(), MRI);
1227 if (!Ext)
1228 return false;
1229
1230 Register ExtSrc1 = Ext->getOperand(1).getReg();
1231 Register ExtSrc2 = Ext->getOperand(2).getReg();
1232 auto LowestVal =
1234 if (!LowestVal || LowestVal->Value.getZExtValue() != DstTy.getSizeInBytes())
1235 return false;
1236
1237 if (!getOpcodeDef<GImplicitDef>(ExtSrc2, MRI))
1238 return false;
1239
1240 MatchInfo = ExtSrc1;
1241 return true;
1242}
1243
1244void applyUnmergeExtToUnmerge(MachineInstr &MI, MachineRegisterInfo &MRI,
1246 GISelChangeObserver &Observer, Register &SrcReg) {
1247 Observer.changingInstr(MI);
1248 // Swap dst registers.
1249 Register Dst1 = MI.getOperand(0).getReg();
1250 MI.getOperand(0).setReg(MI.getOperand(1).getReg());
1251 MI.getOperand(1).setReg(Dst1);
1252 MI.getOperand(2).setReg(SrcReg);
1253 Observer.changedInstr(MI);
1254}
1255
1256// Match mul({z/s}ext , {z/s}ext) => {u/s}mull OR
1257// Match v2s64 mul instructions, which will then be scalarised later on
1258// Doing these two matches in one function to ensure that the order of matching
1259// will always be the same.
1260// Try lowering MUL to MULL before trying to scalarize if needed.
1261bool matchMulv2s64(MachineInstr &MI, MachineRegisterInfo &MRI) {
1262 // Get the instructions that defined the source operand
1263 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
1264 return DstTy == LLT::fixed_vector(2, 64);
1265}
1266
1267void applyMulv2s64(MachineInstr &MI, MachineRegisterInfo &MRI,
1269 assert(MI.getOpcode() == TargetOpcode::G_MUL &&
1270 "Expected a G_MUL instruction");
1271
1272 // Get the instructions that defined the source operand
1273 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
1274 assert(DstTy == LLT::fixed_vector(2, 64) && "Expected v2s64 Mul");
1275 LegalizerHelper Helper(*MI.getMF(), Observer, B);
1276 Helper.fewerElementsVector(
1277 MI, 0,
1279}
1280
1281class AArch64PostLegalizerLoweringImpl : public Combiner {
1282protected:
1283 const CombinerHelper Helper;
1284 const AArch64PostLegalizerLoweringImplRuleConfig &RuleConfig;
1285 const AArch64Subtarget &STI;
1286
1287public:
1288 AArch64PostLegalizerLoweringImpl(
1289 MachineFunction &MF, CombinerInfo &CInfo, GISelCSEInfo *CSEInfo,
1290 const AArch64PostLegalizerLoweringImplRuleConfig &RuleConfig,
1291 const AArch64Subtarget &STI);
1292
1293 static const char *getName() { return "AArch6400PreLegalizerCombiner"; }
1294
1295 bool tryCombineAll(MachineInstr &I) const override;
1296
1297private:
1298#define GET_GICOMBINER_CLASS_MEMBERS
1299#include "AArch64GenPostLegalizeGILowering.inc"
1300#undef GET_GICOMBINER_CLASS_MEMBERS
1301};
1302
1303#define GET_GICOMBINER_IMPL
1304#include "AArch64GenPostLegalizeGILowering.inc"
1305#undef GET_GICOMBINER_IMPL
1306
1307AArch64PostLegalizerLoweringImpl::AArch64PostLegalizerLoweringImpl(
1308 MachineFunction &MF, CombinerInfo &CInfo, GISelCSEInfo *CSEInfo,
1309 const AArch64PostLegalizerLoweringImplRuleConfig &RuleConfig,
1310 const AArch64Subtarget &STI)
1311 : Combiner(MF, CInfo, /*VT*/ nullptr, CSEInfo),
1312 Helper(Observer, B, /*IsPreLegalize*/ true), RuleConfig(RuleConfig),
1313 STI(STI),
1315#include "AArch64GenPostLegalizeGILowering.inc"
1317{
1318}
1319
1320bool runPostLegalizerLowering(
1321 MachineFunction &MF,
1322 const AArch64PostLegalizerLoweringImplRuleConfig &RuleConfig) {
1323 if (MF.getProperties().hasFailedISel())
1324 return false;
1325 const Function &F = MF.getFunction();
1326
1328 CombinerInfo CInfo(/*AllowIllegalOps=*/true, /*ShouldLegalizeIllegal=*/false,
1329 /*LegalizerInfo=*/nullptr, /*OptEnabled=*/true,
1330 F.hasOptSize(), F.hasMinSize());
1331 // Disable fixed-point iteration to reduce compile-time
1332 CInfo.MaxIterations = 1;
1333 CInfo.ObserverLvl = CombinerInfo::ObserverLevel::SinglePass;
1334 // PostLegalizerCombiner performs DCE, so a full DCE pass is unnecessary.
1335 CInfo.EnableFullDCE = false;
1336 AArch64PostLegalizerLoweringImpl Impl(MF, CInfo, /*CSEInfo=*/nullptr,
1337 RuleConfig, ST);
1338 return Impl.combineMachineInstrs();
1339}
1340
1341class AArch64PostLegalizerLoweringLegacy : public MachineFunctionPass {
1342public:
1343 static char ID;
1344
1345 AArch64PostLegalizerLoweringLegacy();
1346
1347 StringRef getPassName() const override {
1348 return "AArch64PostLegalizerLowering";
1349 }
1350
1351 bool runOnMachineFunction(MachineFunction &MF) override;
1352 void getAnalysisUsage(AnalysisUsage &AU) const override;
1353
1354private:
1355 AArch64PostLegalizerLoweringImplRuleConfig RuleConfig;
1356};
1357} // end anonymous namespace
1358
1359void AArch64PostLegalizerLoweringLegacy::getAnalysisUsage(
1360 AnalysisUsage &AU) const {
1361 AU.setPreservesCFG();
1364}
1365
1366AArch64PostLegalizerLoweringLegacy::AArch64PostLegalizerLoweringLegacy()
1367 : MachineFunctionPass(ID) {
1368 if (!RuleConfig.parseCommandLineOption())
1369 report_fatal_error("Invalid rule identifier");
1370}
1371
1372bool AArch64PostLegalizerLoweringLegacy::runOnMachineFunction(
1373 MachineFunction &MF) {
1374 assert(MF.getProperties().hasLegalized() && "Expected a legalized function?");
1375 return runPostLegalizerLowering(MF, RuleConfig);
1376}
1377
1378char AArch64PostLegalizerLoweringLegacy::ID = 0;
1379INITIALIZE_PASS_BEGIN(AArch64PostLegalizerLoweringLegacy, DEBUG_TYPE,
1380 "Lower AArch64 MachineInstrs after legalization", false,
1381 false)
1382INITIALIZE_PASS_END(AArch64PostLegalizerLoweringLegacy, DEBUG_TYPE,
1383 "Lower AArch64 MachineInstrs after legalization", false,
1384 false)
1385
1387 : RuleConfig(
1388 std::make_unique<AArch64PostLegalizerLoweringImplRuleConfig>()) {
1389 if (!RuleConfig->parseCommandLineOption())
1390 reportFatalUsageError("invalid rule identifier");
1391}
1392
1395
1397
1401 MFPropsModifier _(*this, MF);
1402 const bool Changed = runPostLegalizerLowering(MF, *RuleConfig);
1403
1404 if (!Changed)
1405 return PreservedAnalyses::all();
1406
1409 return PA;
1410}
1411
1412namespace llvm {
1414 return new AArch64PostLegalizerLoweringLegacy();
1415}
1416} // end namespace llvm
static bool isVShiftRImm(SDValue Op, EVT VT, bool isNarrow, int64_t &Cnt)
isVShiftRImm - Check if this is a valid build_vector for the immediate operand of a vector shift righ...
static bool isINSMask(ArrayRef< int > M, int NumInputElements, bool &DstIsLeft, int &Anomaly)
static unsigned getCmpOperandFoldingProfit(SDValue Op, bool AllowExtend)
Returns how profitable it is to fold a comparison's operand's shift and/or extension operations.
static bool shouldBeAdjustedToZero(SDValue LHS, const APInt &C, ISD::CondCode &CC)
This file declares the targeting of the Machinelegalizer class for AArch64.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
unsigned Imm
unsigned uint64_t
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
This contains common code to allow clients to notify changes to machine instr.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
#define _
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineIRBuilder class.
Register Reg
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
#define P(N)
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
static StringRef getName(Value *V)
Value * RHS
Value * LHS
BinaryOperator * Mul
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
unsigned logBase2() const
Definition APInt.h:1782
Represent the analysis usage information of a pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
@ ICMP_SLE
signed less or equal
Definition InstrTypes.h:770
@ ICMP_UGE
unsigned greater or equal
Definition InstrTypes.h:764
@ ICMP_UGT
unsigned greater than
Definition InstrTypes.h:763
@ ICMP_SGT
signed greater than
Definition InstrTypes.h:767
@ ICMP_ULT
unsigned less than
Definition InstrTypes.h:765
@ FCMP_ORD
0 1 1 1 True if ordered (no nans)
Definition InstrTypes.h:749
@ ICMP_SGE
signed greater or equal
Definition InstrTypes.h:768
@ ICMP_ULE
unsigned less or equal
Definition InstrTypes.h:766
@ FCMP_UNO
1 0 0 0 True if unordered: isnan(X) | isnan(Y)
Definition InstrTypes.h:750
Predicate getSwappedPredicate() const
For example, EQ->EQ, SLE->SGE, ULT->UGT, OEQ->OEQ, ULE->UGE, OLT->OGT, etc.
Definition InstrTypes.h:890
Combiner implementation.
Definition Combiner.h:33
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
The CSE Analysis object.
Definition CSEInfo.h:72
Abstract class that contains various methods for clients to notify about changes.
virtual void changingInstr(MachineInstr &MI)=0
This instruction is about to be mutated in some way.
virtual void changedInstr(MachineInstr &MI)=0
This instruction was mutated in some way.
LLT changeElementCount(ElementCount EC) const
Return a vector or scalar with the same element type and the new element count.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr uint16_t getNumElements() const
Returns the number of elements in a vector LLT.
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr ElementCount getElementCount() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
static LLT integer(unsigned SizeInBits)
constexpr TypeSize getSizeInBytes() const
Returns the total size of the type in bytes, i.e.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
LLVM_ABI LegalizeResult lower(MachineInstr &MI, unsigned TypeIdx, LLT Ty)
Legalize an instruction by splitting it into simpler parts, hopefully understood by the target.
LLVM_ABI LegalizeResult fewerElementsVector(MachineInstr &MI, unsigned TypeIdx, LLT NarrowTy)
Legalize a vector instruction by splitting into multiple components, each acting on the same scalar t...
An RAII based helper class to modify MachineFunctionProperties when running pass.
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
Helper class to build MachineInstr.
MachineInstrBuilder buildNot(const DstOp &Dst, const SrcOp &Src0)
Build and insert a bitwise not, NegOne = G_CONSTANT -1 Res = G_OR Op0, NegOne.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
void setInstrAndDebugLoc(MachineInstr &MI)
Set the insertion point to before MI, and set the debug loc to MI's loc.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineInstrBuilder buildOr(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_OR Op0, Op1.
MachineInstrBuilder buildCopy(const DstOp &Res, const SrcOp &Op)
Build and insert Res = COPY Op.
virtual MachineInstrBuilder buildConstant(const DstOp &Res, const ConstantInt &Val)
Build and insert Res = G_CONSTANT Val.
Register getReg(unsigned Idx) const
Get the register for the operand index.
Representation of each machine instruction.
const MachineOperand & getOperand(unsigned i) const
MachineOperand class - Representation of each machine instruction operand.
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
bool use_nodbg_empty(Register RegNo) const
use_nodbg_empty - Return true if there are no non-Debug instructions using the specified register.
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
Wrapper class representing virtual and physical registers.
Definition Register.h:20
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
void changeVectorFCMPPredToAArch64CC(const CmpInst::Predicate P, AArch64CC::CondCode &CondCode, AArch64CC::CondCode &CondCode2, bool &Invert)
Find the AArch64 condition codes necessary to represent P for a vector floating point comparison.
bool isCMN(const MachineInstr *MaybeSub, const CmpInst::Predicate &Pred, const MachineRegisterInfo &MRI)
std::optional< int64_t > getAArch64VectorSplatScalar(const MachineInstr &MI, const MachineRegisterInfo &MRI)
static bool isLegalCmpImmed(const APInt &C)
isLegalCmpImmed -
void expandMOVImm(uint64_t Imm, unsigned BitSize, SmallVectorImpl< ImmInsnModel > &Insn)
Expand a MOVi32imm or MOVi64imm pseudo instruction to one or more real move-immediate instructions to...
operand_type_match m_Reg()
SpecificConstantMatch m_ZeroInt()
Convenience matchers for specific integer values.
ImplicitDefMatch m_GImplicitDef()
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
UnaryOp_match< SrcTy, TargetOpcode::G_TRUNC > m_GTrunc(const SrcTy &Src)
This is an optimization pass for GlobalISel generic memory operations.
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:315
@ Offset
Definition DWP.cpp:577
LLVM_ABI bool isBuildVectorAllZeros(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndef=false)
Return true if the specified instruction is a G_BUILD_VECTOR or G_BUILD_VECTOR_TRUNC where all of the...
Definition Utils.cpp:1434
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
Definition Utils.cpp:656
bool isZIPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for zip1 or zip2 masks of the form: <0, 8, 1, 9, 2, 10, 3, 11> (WhichResultOut = 0,...
@ Undef
Value of the register doesn't matter.
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
Definition Utils.cpp:497
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
FunctionPass * createAArch64PostLegalizerLowering()
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
bool isUZPMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut)
Return true for uzp1 or uzp2 masks of the form: <0, 2, 4, 6, 8, 10, 12, 14> or <1,...
bool isREVMask(ArrayRef< int > M, unsigned EltSize, unsigned NumElts, unsigned BlockSize)
isREVMask - Check if a vector shuffle corresponds to a REV instruction with the specified blocksize.
bool isUZP_v_undef_Mask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResult)
isUZP_v_undef_Mask - Special case of isUZPMask for canonical form of "vector_shuffle v,...
LLVM_ABI std::optional< ValueAndVReg > getAnyConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true, bool LookThroughAnyExt=false)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT or G_FCONST...
Definition Utils.cpp:442
LLVM_ABI bool isBuildVectorAllOnes(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndef=false)
Return true if the specified instruction is a G_BUILD_VECTOR or G_BUILD_VECTOR_TRUNC where all of the...
Definition Utils.cpp:1440
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
Definition Utils.cpp:1137
bool isZIP_v_undef_Mask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResult)
isZIP_v_undef_Mask - Special case of isZIPMask for canonical form of "vector_shuffle v,...
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
bool isTRN_v_undef_Mask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResult)
isTRN_v_undef_Mask - Special case of isTRNMask for canonical form of "vector_shuffle v,...
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1772
bool isTRNMask(ArrayRef< int > M, unsigned NumElts, unsigned &WhichResultOut, unsigned &OperandOrderOut)
Return true for trn1 or trn2 masks of the form: <0, 8, 2, 10, 4, 12, 6, 14> (WhichResultOut = 0,...
LLVM_ABI int getSplatIndex(ArrayRef< int > Mask)
If all non-negative Mask elements are the same value, return that value.
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
@ SinglePass
Enables Observer-based DCE and additional heuristics that retry combining defined and used instructio...
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.