LLVM 24.0.0git
AArch64PostLegalizerCombiner.cpp
Go to the documentation of this file.
1//=== AArch64PostLegalizerCombiner.cpp --------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8///
9/// \file
10/// Post-legalization combines on generic MachineInstrs.
11///
12/// The combines here must preserve instruction legality.
13///
14/// Lowering combines (e.g. pseudo matching) should be handled by
15/// AArch64PostLegalizerLowering.
16///
17/// Combines which don't rely on instruction legality should go in the
18/// AArch64PreLegalizerCombiner.
19///
20//===----------------------------------------------------------------------===//
21
22#include "AArch64.h"
24#include "llvm/ADT/STLExtras.h"
43#include "llvm/Support/Debug.h"
44
45#define GET_GICOMBINER_DEPS
46#include "AArch64GenPostLegalizeGICombiner.inc"
47#undef GET_GICOMBINER_DEPS
48
49#define DEBUG_TYPE "aarch64-postlegalizer-combiner"
50
51using namespace llvm;
52using namespace MIPatternMatch;
53
54#define GET_GICOMBINER_TYPES
55#include "AArch64GenPostLegalizeGICombiner.inc"
56#undef GET_GICOMBINER_TYPES
57
58namespace {
59
60/// This combine tries do what performExtractVectorEltCombine does in SDAG.
61/// Rewrite for pairwise fadd pattern
62/// (s32 (g_extract_vector_elt
63/// (g_fadd (vXs32 Other)
64/// (g_vector_shuffle (vXs32 Other) undef <1,X,...> )) 0))
65/// ->
66/// (s32 (g_fadd (g_extract_vector_elt (vXs32 Other) 0)
67/// (g_extract_vector_elt (vXs32 Other) 1))
68bool matchExtractVecEltPairwiseAdd(
70 std::tuple<unsigned, LLT, Register> &MatchInfo) {
71 Register Src1 = MI.getOperand(1).getReg();
72 Register Src2 = MI.getOperand(2).getReg();
73 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
74
75 auto Cst = getIConstantVRegValWithLookThrough(Src2, MRI);
76 if (!Cst || Cst->Value != 0)
77 return false;
78 // SDAG also checks for FullFP16, but this looks to be beneficial anyway.
79
80 // Now check for an fadd operation. TODO: expand this for integer add?
81 auto *FAddMI = getOpcodeDef(TargetOpcode::G_FADD, Src1, MRI);
82 if (!FAddMI)
83 return false;
84
85 // If we add support for integer add, must restrict these types to just s64.
86 unsigned DstSize = DstTy.getSizeInBits();
87 if (DstSize != 16 && DstSize != 32 && DstSize != 64)
88 return false;
89
90 Register Src1Op1 = FAddMI->getOperand(1).getReg();
91 Register Src1Op2 = FAddMI->getOperand(2).getReg();
92 MachineInstr *Shuffle =
93 getOpcodeDef(TargetOpcode::G_SHUFFLE_VECTOR, Src1Op2, MRI);
94 MachineInstr *Other = MRI.getVRegDef(Src1Op1);
95 if (!Shuffle) {
96 Shuffle = getOpcodeDef(TargetOpcode::G_SHUFFLE_VECTOR, Src1Op1, MRI);
97 Other = MRI.getVRegDef(Src1Op2);
98 }
99
100 // We're looking for a shuffle that moves the second element to index 0.
101 if (Shuffle && Shuffle->getOperand(3).getShuffleMask()[0] == 1 &&
102 Other == MRI.getVRegDef(Shuffle->getOperand(1).getReg())) {
103 std::get<0>(MatchInfo) = TargetOpcode::G_FADD;
104 std::get<1>(MatchInfo) = DstTy;
105 std::get<2>(MatchInfo) = Other->getOperand(0).getReg();
106 return true;
107 }
108 return false;
109}
110
111void applyExtractVecEltPairwiseAdd(
113 std::tuple<unsigned, LLT, Register> &MatchInfo) {
114 unsigned Opc = std::get<0>(MatchInfo);
115 assert(Opc == TargetOpcode::G_FADD && "Unexpected opcode!");
116 // We want to generate two extracts of elements 0 and 1, and add them.
117 LLT Ty = std::get<1>(MatchInfo);
118 Register Src = std::get<2>(MatchInfo);
119 LLT s64 = LLT::integer(64);
120 B.setInstrAndDebugLoc(MI);
121 auto Elt0 = B.buildExtractVectorElement(Ty, Src, B.buildConstant(s64, 0));
122 auto Elt1 = B.buildExtractVectorElement(Ty, Src, B.buildConstant(s64, 1));
123 B.buildInstr(Opc, {MI.getOperand(0).getReg()}, {Elt0, Elt1});
124 MI.eraseFromParent();
125}
126
128 // TODO: check if extended build vector as well.
129 return mi_match(R, MRI, m_GSExt(m_Reg())) ||
130 mi_match(R, MRI, m_GSExtInReg(m_Reg()));
131}
132
134 // TODO: check if extended build vector as well.
135 return mi_match(R, MRI, m_GZExt(m_Reg()));
136}
137
138bool matchAArch64MulConstCombine(
140 std::function<void(MachineIRBuilder &B, Register DstReg)> &ApplyFn) {
141 assert(MI.getOpcode() == TargetOpcode::G_MUL);
142 Register LHS = MI.getOperand(1).getReg();
143 Register RHS = MI.getOperand(2).getReg();
144 Register Dst = MI.getOperand(0).getReg();
145 const LLT Ty = MRI.getType(LHS);
146
147 // The below optimizations require a constant RHS.
148 auto Const = getIConstantVRegValWithLookThrough(RHS, MRI);
149 if (!Const)
150 return false;
151
152 APInt ConstValue = Const->Value.sext(Ty.getSizeInBits());
153 // The following code is ported from AArch64ISelLowering.
154 // Multiplication of a power of two plus/minus one can be done more
155 // cheaply as shift+add/sub. For now, this is true unilaterally. If
156 // future CPUs have a cheaper MADD instruction, this may need to be
157 // gated on a subtarget feature. For Cyclone, 32-bit MADD is 4 cycles and
158 // 64-bit is 5 cycles, so this is always a win.
159 // More aggressively, some multiplications N0 * C can be lowered to
160 // shift+add+shift if the constant C = A * B where A = 2^N + 1 and B = 2^M,
161 // e.g. 6=3*2=(2+1)*2.
162 // TODO: consider lowering more cases, e.g. C = 14, -6, -14 or even 45
163 // which equals to (1+2)*16-(1+2).
164 // TrailingZeroes is used to test if the mul can be lowered to
165 // shift+add+shift.
166 unsigned TrailingZeroes = ConstValue.countr_zero();
167 if (TrailingZeroes) {
168 // Conservatively do not lower to shift+add+shift if the mul might be
169 // folded into smul or umul.
170 if (MRI.hasOneNonDBGUse(LHS) &&
171 (isSignExtended(LHS, MRI) || isZeroExtended(LHS, MRI)))
172 return false;
173 // Conservatively do not lower to shift+add+shift if the mul might be
174 // folded into madd or msub.
175 if (MRI.hasOneNonDBGUse(Dst)) {
177 unsigned UseOpc = UseMI.getOpcode();
178 if (UseOpc == TargetOpcode::G_ADD || UseOpc == TargetOpcode::G_PTR_ADD ||
179 UseOpc == TargetOpcode::G_SUB)
180 return false;
181 }
182 }
183 // Use ShiftedConstValue instead of ConstValue to support both shift+add/sub
184 // and shift+add+shift.
185 APInt ShiftedConstValue = ConstValue.ashr(TrailingZeroes);
186
187 unsigned ShiftAmt, AddSubOpc;
188 // Is the shifted value the LHS operand of the add/sub?
189 bool ShiftValUseIsLHS = true;
190 // Do we need to negate the result?
191 bool NegateResult = false;
192
193 if (ConstValue.isNonNegative()) {
194 // (mul x, 2^N + 1) => (add (shl x, N), x)
195 // (mul x, 2^N - 1) => (sub (shl x, N), x)
196 // (mul x, (2^N + 1) * 2^M) => (shl (add (shl x, N), x), M)
197 APInt SCVMinus1 = ShiftedConstValue - 1;
198 APInt CVPlus1 = ConstValue + 1;
199 if (SCVMinus1.isPowerOf2()) {
200 ShiftAmt = SCVMinus1.logBase2();
201 AddSubOpc = TargetOpcode::G_ADD;
202 } else if (CVPlus1.isPowerOf2()) {
203 ShiftAmt = CVPlus1.logBase2();
204 AddSubOpc = TargetOpcode::G_SUB;
205 } else
206 return false;
207 } else {
208 // (mul x, -(2^N - 1)) => (sub x, (shl x, N))
209 // (mul x, -(2^N + 1)) => - (add (shl x, N), x)
210 APInt CVNegPlus1 = -ConstValue + 1;
211 APInt CVNegMinus1 = -ConstValue - 1;
212 if (CVNegPlus1.isPowerOf2()) {
213 ShiftAmt = CVNegPlus1.logBase2();
214 AddSubOpc = TargetOpcode::G_SUB;
215 ShiftValUseIsLHS = false;
216 } else if (CVNegMinus1.isPowerOf2()) {
217 ShiftAmt = CVNegMinus1.logBase2();
218 AddSubOpc = TargetOpcode::G_ADD;
219 NegateResult = true;
220 } else
221 return false;
222 }
223
224 if (NegateResult && TrailingZeroes)
225 return false;
226
227 ApplyFn = [=](MachineIRBuilder &B, Register DstReg) {
228 auto Shift = B.buildConstant(LLT::integer(64), ShiftAmt);
229 auto ShiftedVal = B.buildShl(Ty, LHS, Shift);
230
231 Register AddSubLHS = ShiftValUseIsLHS ? ShiftedVal.getReg(0) : LHS;
232 Register AddSubRHS = ShiftValUseIsLHS ? LHS : ShiftedVal.getReg(0);
233 auto Res = B.buildInstr(AddSubOpc, {Ty}, {AddSubLHS, AddSubRHS});
234 assert(!(NegateResult && TrailingZeroes) &&
235 "NegateResult and TrailingZeroes cannot both be true for now.");
236 // Negate the result.
237 if (NegateResult) {
238 B.buildSub(DstReg, B.buildConstant(Ty, 0), Res);
239 return;
240 }
241 // Shift the result.
242 if (TrailingZeroes) {
243 B.buildShl(DstReg, Res,
244 B.buildConstant(LLT::integer(64), TrailingZeroes));
245 return;
246 }
247 B.buildCopy(DstReg, Res.getReg(0));
248 };
249 return true;
250}
251
252void applyAArch64MulConstCombine(
254 std::function<void(MachineIRBuilder &B, Register DstReg)> &ApplyFn) {
255 B.setInstrAndDebugLoc(MI);
256 ApplyFn(B, MI.getOperand(0).getReg());
257 MI.eraseFromParent();
258}
259
260/// Match a 128b store of zero and split it into two 64 bit stores, for
261/// size/performance reasons.
262bool matchSplitStoreZero128(MachineInstr &MI, MachineRegisterInfo &MRI) {
264 if (!Store.isSimple())
265 return false;
266 LLT ValTy = MRI.getType(Store.getValueReg());
267 if (ValTy.isScalableVector())
268 return false;
269 if (!ValTy.isVector() || ValTy.getSizeInBits() != 128)
270 return false;
271 if (Store.getMemSizeInBits() != ValTy.getSizeInBits())
272 return false; // Don't split truncating stores.
273 if (!MRI.hasOneNonDBGUse(Store.getValueReg()))
274 return false;
275 auto MaybeCst = isConstantOrConstantSplatVector(Store.getValueReg(), MRI);
276 return MaybeCst && MaybeCst->isZero();
277}
278
279void applySplitStoreZero128(MachineInstr &MI, MachineRegisterInfo &MRI,
281 GISelChangeObserver &Observer) {
282 B.setInstrAndDebugLoc(MI);
284 assert(MRI.getType(Store.getValueReg()).isVector() &&
285 "Expected a vector store value");
286 LLT NewTy = LLT::integer(64);
287 Register PtrReg = Store.getPointerReg();
288 auto Zero = B.buildConstant(NewTy, 0);
289 auto HighPtr =
290 B.buildPtrAdd(MRI.getType(PtrReg), PtrReg, B.buildConstant(NewTy, 8));
291 auto &MF = *MI.getMF();
292 auto *LowMMO = MF.getMachineMemOperand(&Store.getMMO(), 0, NewTy);
293 auto *HighMMO = MF.getMachineMemOperand(&Store.getMMO(), 8, NewTy);
294 B.buildStore(Zero, PtrReg, *LowMMO);
295 B.buildStore(Zero, HighPtr, *HighMMO);
296 Store.eraseFromParent();
297}
298
299bool matchOrToBSP(MachineInstr &MI, MachineRegisterInfo &MRI,
300 std::tuple<Register, Register, Register> &MatchInfo) {
301 const LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
302 if (!DstTy.isVector())
303 return false;
304
305 Register AO1, AO2, BVO1, BVO2;
306 if (!mi_match(MI, MRI,
307 m_GOr(m_GAnd(m_Reg(AO1), m_Reg(BVO1)),
308 m_GAnd(m_Reg(AO2), m_Reg(BVO2)))))
309 return false;
310
311 auto *BV1 = getOpcodeDef<GBuildVector>(BVO1, MRI);
312 auto *BV2 = getOpcodeDef<GBuildVector>(BVO2, MRI);
313 if (!BV1 || !BV2)
314 return false;
315
316 for (int I = 0, E = DstTy.getNumElements(); I < E; I++) {
317 auto ValAndVReg1 =
318 getIConstantVRegValWithLookThrough(BV1->getSourceReg(I), MRI);
319 auto ValAndVReg2 =
320 getIConstantVRegValWithLookThrough(BV2->getSourceReg(I), MRI);
321 if (!ValAndVReg1 || !ValAndVReg2 ||
322 ValAndVReg1->Value != ~ValAndVReg2->Value)
323 return false;
324 }
325
326 MatchInfo = {AO1, AO2, BVO1};
327 return true;
328}
329
330void applyOrToBSP(MachineInstr &MI, MachineRegisterInfo &MRI,
332 std::tuple<Register, Register, Register> &MatchInfo) {
333 B.setInstrAndDebugLoc(MI);
334 B.buildInstr(
335 AArch64::G_BSP, {MI.getOperand(0).getReg()},
336 {std::get<2>(MatchInfo), std::get<0>(MatchInfo), std::get<1>(MatchInfo)});
337 MI.eraseFromParent();
338}
339
340/// Match G_TRUNC (G_OR X, Y) => G_ADDHN X, Y when both inputs are sign
341/// extended from the result element type. The high half of the addition then
342/// equals the truncation of the OR.
343bool matchTruncOrToADDHN(MachineInstr &MI, MachineRegisterInfo &MRI,
345 Register Src0, Register Src1) {
346 if (!MRI.hasOneUse(Or))
347 return false;
348
349 LLT DstTy = MRI.getType(Dst);
350 LLT SrcTy = MRI.getType(Or);
351 if (!((DstTy == LLT::fixed_vector(8, 8) &&
352 SrcTy == LLT::fixed_vector(8, 16)) ||
353 (DstTy == LLT::fixed_vector(4, 16) &&
354 SrcTy == LLT::fixed_vector(4, 32)) ||
355 (DstTy == LLT::fixed_vector(2, 32) &&
356 SrcTy == LLT::fixed_vector(2, 64))))
357 return false;
358
359 // If the narrow result is immediately any-extended back to the original type,
360 // the G_OR is cheaper than G_ADDHN followed by a vector widen.
361 if (MRI.hasOneNonDBGUse(Dst)) {
362 MachineInstr &UseMI = *MRI.use_nodbg_instructions(Dst).begin();
363 if (UseMI.getOpcode() == TargetOpcode::G_ANYEXT &&
364 MRI.getType(UseMI.getOperand(0).getReg()) == SrcTy)
365 return false;
366 }
367
368 unsigned EltSize = SrcTy.getScalarSizeInBits();
369 if (VT->computeNumSignBits(Src0) != EltSize ||
370 VT->computeNumSignBits(Src1) != EltSize)
371 return false;
372
373 return true;
374}
375
376// Combines Mul(And(Srl(X, 15), 0x10001), 0xffff) into CMLTz
377bool matchCombineMulCMLT(MachineInstr &MI, MachineRegisterInfo &MRI,
378 Register &SrcReg) {
379 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
380
381 if (DstTy != LLT::fixed_vector(2, 64) && DstTy != LLT::fixed_vector(2, 32) &&
382 DstTy != LLT::fixed_vector(4, 32) && DstTy != LLT::fixed_vector(4, 16) &&
383 DstTy != LLT::fixed_vector(8, 16))
384 return false;
385
386 auto AndMI = getDefIgnoringCopies(MI.getOperand(1).getReg(), MRI);
387 if (AndMI->getOpcode() != TargetOpcode::G_AND)
388 return false;
389 auto LShrMI = getDefIgnoringCopies(AndMI->getOperand(1).getReg(), MRI);
390 if (LShrMI->getOpcode() != TargetOpcode::G_LSHR)
391 return false;
392
393 // Check the constant splat values
394 auto V1 = isConstantOrConstantSplatVector(MI.getOperand(2).getReg(), MRI);
395 auto V2 = isConstantOrConstantSplatVector(AndMI->getOperand(2).getReg(), MRI);
396 auto V3 =
397 isConstantOrConstantSplatVector(LShrMI->getOperand(2).getReg(), MRI);
398 if (!V1.has_value() || !V2.has_value() || !V3.has_value())
399 return false;
400 unsigned HalfSize = DstTy.getScalarSizeInBits() / 2;
401 if (!V1.value().isMask(HalfSize) || V2.value() != (1ULL | 1ULL << HalfSize) ||
402 V3 != (HalfSize - 1))
403 return false;
404
405 SrcReg = LShrMI->getOperand(1).getReg();
406
407 return true;
408}
409
410void applyCombineMulCMLT(MachineInstr &MI, MachineRegisterInfo &MRI,
411 MachineIRBuilder &B, Register &SrcReg) {
412 Register DstReg = MI.getOperand(0).getReg();
413 LLT DstTy = MRI.getType(DstReg);
414 LLT HalfTy = DstTy.changeElementCount(DstTy.getElementCount() * 2)
416
417 Register ZeroVec = B.buildConstant(HalfTy, 0).getReg(0);
418 Register CastReg =
419 B.buildInstr(TargetOpcode::G_BITCAST, {HalfTy}, {SrcReg}).getReg(0);
420 Register CMLTReg =
421 B.buildICmp(CmpInst::Predicate::ICMP_SLT, HalfTy, CastReg, ZeroVec)
422 .getReg(0);
423
424 B.buildInstr(TargetOpcode::G_BITCAST, {DstReg}, {CMLTReg}).getReg(0);
425 MI.eraseFromParent();
426}
427
428// Match mul({z/s}ext , {z/s}ext) => {u/s}mull
429bool matchExtMulToMULL(MachineInstr &MI, MachineRegisterInfo &MRI,
431 std::tuple<bool, Register, Register> &MatchInfo) {
432 // Get the instructions that defined the source operand
433 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
434 MachineInstr *I1 = getDefIgnoringCopies(MI.getOperand(1).getReg(), MRI);
435 MachineInstr *I2 = getDefIgnoringCopies(MI.getOperand(2).getReg(), MRI);
436 unsigned I1Opc = I1->getOpcode();
437 unsigned I2Opc = I2->getOpcode();
438 unsigned EltSize = DstTy.getScalarSizeInBits();
439
440 if (!DstTy.isVector() || I1->getNumOperands() < 2 || I2->getNumOperands() < 2)
441 return false;
442
443 auto IsAtLeastDoubleExtend = [&](Register R) {
444 LLT Ty = MRI.getType(R);
445 return EltSize >= Ty.getScalarSizeInBits() * 2;
446 };
447
448 // If the source operands were EXTENDED before, then {U/S}MULL can be used
449 bool IsZExt1 =
450 I1Opc == TargetOpcode::G_ZEXT || I1Opc == TargetOpcode::G_ANYEXT;
451 bool IsZExt2 =
452 I2Opc == TargetOpcode::G_ZEXT || I2Opc == TargetOpcode::G_ANYEXT;
453 if (IsZExt1 && IsZExt2 && IsAtLeastDoubleExtend(I1->getOperand(1).getReg()) &&
454 IsAtLeastDoubleExtend(I2->getOperand(1).getReg())) {
455 get<0>(MatchInfo) = true;
456 get<1>(MatchInfo) = I1->getOperand(1).getReg();
457 get<2>(MatchInfo) = I2->getOperand(1).getReg();
458 return true;
459 }
460
461 bool IsSExt1 =
462 I1Opc == TargetOpcode::G_SEXT || I1Opc == TargetOpcode::G_ANYEXT;
463 bool IsSExt2 =
464 I2Opc == TargetOpcode::G_SEXT || I2Opc == TargetOpcode::G_ANYEXT;
465 if (IsSExt1 && IsSExt2 && IsAtLeastDoubleExtend(I1->getOperand(1).getReg()) &&
466 IsAtLeastDoubleExtend(I2->getOperand(1).getReg())) {
467 get<0>(MatchInfo) = false;
468 get<1>(MatchInfo) = I1->getOperand(1).getReg();
469 get<2>(MatchInfo) = I2->getOperand(1).getReg();
470 return true;
471 }
472
473 // Select UMULL if we can replace the other operand with an extend.
474 APInt Mask = APInt::getHighBitsSet(EltSize, EltSize / 2);
475 if (KB && (IsZExt1 || IsZExt2) &&
476 IsAtLeastDoubleExtend(IsZExt1 ? I1->getOperand(1).getReg()
477 : I2->getOperand(1).getReg())) {
478 Register ZExtOp =
479 IsZExt1 ? MI.getOperand(2).getReg() : MI.getOperand(1).getReg();
480 if (KB->maskedValueIsZero(ZExtOp, Mask)) {
481 get<0>(MatchInfo) = true;
482 get<1>(MatchInfo) = IsZExt1 ? I1->getOperand(1).getReg() : ZExtOp;
483 get<2>(MatchInfo) = IsZExt1 ? ZExtOp : I2->getOperand(1).getReg();
484 return true;
485 }
486 } else if (KB && DstTy == LLT::fixed_vector(2, 64) &&
487 KB->maskedValueIsZero(MI.getOperand(1).getReg(), Mask) &&
488 KB->maskedValueIsZero(MI.getOperand(2).getReg(), Mask)) {
489 get<0>(MatchInfo) = true;
490 get<1>(MatchInfo) = MI.getOperand(1).getReg();
491 get<2>(MatchInfo) = MI.getOperand(2).getReg();
492 return true;
493 }
494
495 if (KB && (IsSExt1 || IsSExt2) &&
496 IsAtLeastDoubleExtend(IsSExt1 ? I1->getOperand(1).getReg()
497 : I2->getOperand(1).getReg())) {
498 Register SExtOp =
499 IsSExt1 ? MI.getOperand(2).getReg() : MI.getOperand(1).getReg();
500 if (KB->computeNumSignBits(SExtOp) > EltSize / 2) {
501 get<0>(MatchInfo) = false;
502 get<1>(MatchInfo) = IsSExt1 ? I1->getOperand(1).getReg() : SExtOp;
503 get<2>(MatchInfo) = IsSExt1 ? SExtOp : I2->getOperand(1).getReg();
504 return true;
505 }
506 } else if (KB && DstTy == LLT::fixed_vector(2, 64) &&
507 KB->computeNumSignBits(MI.getOperand(1).getReg()) > EltSize / 2 &&
508 KB->computeNumSignBits(MI.getOperand(2).getReg()) > EltSize / 2) {
509 get<0>(MatchInfo) = false;
510 get<1>(MatchInfo) = MI.getOperand(1).getReg();
511 get<2>(MatchInfo) = MI.getOperand(2).getReg();
512 return true;
513 }
514
515 return false;
516}
517
518void applyExtMulToMULL(MachineInstr &MI, MachineRegisterInfo &MRI,
520 std::tuple<bool, Register, Register> &MatchInfo) {
521 assert(MI.getOpcode() == TargetOpcode::G_MUL &&
522 "Expected a G_MUL instruction");
523
524 // Get the instructions that defined the source operand
525 LLT DstTy = MRI.getType(MI.getOperand(0).getReg());
526 bool IsZExt = get<0>(MatchInfo);
527 Register Src1Reg = get<1>(MatchInfo);
528 Register Src2Reg = get<2>(MatchInfo);
529 LLT Src1Ty = MRI.getType(Src1Reg);
530 LLT Src2Ty = MRI.getType(Src2Reg);
531 LLT HalfDstTy = DstTy.changeElementSize(DstTy.getScalarSizeInBits() / 2);
532 unsigned ExtOpc = IsZExt ? TargetOpcode::G_ZEXT : TargetOpcode::G_SEXT;
533
534 if (Src1Ty.getScalarSizeInBits() * 2 != DstTy.getScalarSizeInBits())
535 Src1Reg = B.buildExtOrTrunc(ExtOpc, {HalfDstTy}, {Src1Reg}).getReg(0);
536 if (Src2Ty.getScalarSizeInBits() * 2 != DstTy.getScalarSizeInBits())
537 Src2Reg = B.buildExtOrTrunc(ExtOpc, {HalfDstTy}, {Src2Reg}).getReg(0);
538
539 B.buildInstr(IsZExt ? AArch64::G_UMULL : AArch64::G_SMULL,
540 {MI.getOperand(0).getReg()}, {Src1Reg, Src2Reg});
541 MI.eraseFromParent();
542}
543
544static bool matchSubAddMulReassoc(Register Mul1, Register Mul2, Register Sub,
545 Register Src, MachineRegisterInfo &MRI) {
546 if (!MRI.hasOneUse(Sub))
547 return false;
549 return false;
551 if (M1->getOpcode() != AArch64::G_MUL &&
552 M1->getOpcode() != AArch64::G_SMULL &&
553 M1->getOpcode() != AArch64::G_UMULL)
554 return false;
555 MachineInstr *M2 = getDefIgnoringCopies(Mul2, MRI);
556 if (M2->getOpcode() != AArch64::G_MUL &&
557 M2->getOpcode() != AArch64::G_SMULL &&
558 M2->getOpcode() != AArch64::G_UMULL)
559 return false;
560 return true;
561}
562
563class AArch64PostLegalizerCombinerImpl : public Combiner {
564protected:
565 const CombinerHelper Helper;
566 const AArch64PostLegalizerCombinerImplRuleConfig &RuleConfig;
567 const AArch64Subtarget &STI;
568
569public:
570 AArch64PostLegalizerCombinerImpl(
572 GISelCSEInfo *CSEInfo,
573 const AArch64PostLegalizerCombinerImplRuleConfig &RuleConfig,
574 const AArch64Subtarget &STI, MachineDominatorTree *MDT,
575 const LegalizerInfo *LI);
576
577 static const char *getName() { return "AArch64PostLegalizerCombiner"; }
578
579 bool tryCombineAll(MachineInstr &I) const override;
580
581private:
582#define GET_GICOMBINER_CLASS_MEMBERS
583#include "AArch64GenPostLegalizeGICombiner.inc"
584#undef GET_GICOMBINER_CLASS_MEMBERS
585};
586
587#define GET_GICOMBINER_IMPL
588#include "AArch64GenPostLegalizeGICombiner.inc"
589#undef GET_GICOMBINER_IMPL
590
591AArch64PostLegalizerCombinerImpl::AArch64PostLegalizerCombinerImpl(
593 GISelCSEInfo *CSEInfo,
594 const AArch64PostLegalizerCombinerImplRuleConfig &RuleConfig,
595 const AArch64Subtarget &STI, MachineDominatorTree *MDT,
596 const LegalizerInfo *LI)
597 : Combiner(MF, CInfo, &VT, CSEInfo),
598 Helper(Observer, B, /*IsPreLegalize*/ false, &VT, MDT, LI),
599 RuleConfig(RuleConfig), STI(STI),
601#include "AArch64GenPostLegalizeGICombiner.inc"
603{
604}
605
606struct StoreInfo {
607 GStore *St = nullptr;
608 // The G_PTR_ADD that's used by the store. We keep this to cache the
609 // MachineInstr def.
610 GPtrAdd *Ptr = nullptr;
611 // The signed offset to the Ptr instruction.
612 int64_t Offset = 0;
613 LLT StoredType;
614};
615
616static bool tryOptimizeConsecStores(SmallVectorImpl<StoreInfo> &Stores,
617 CSEMIRBuilder &MIB) {
618 if (Stores.size() <= 2)
619 return false;
620
621 // Profitabity checks:
622 int64_t BaseOffset = Stores[0].Offset;
623 unsigned NumPairsExpected = Stores.size() / 2;
624 unsigned TotalInstsExpected = NumPairsExpected + (Stores.size() % 2);
625 // Size savings will depend on whether we can fold the offset, as an
626 // immediate of an ADD.
627 auto &TLI = *MIB.getMF().getSubtarget().getTargetLowering();
628 if (!TLI.isLegalAddImmediate(BaseOffset))
629 TotalInstsExpected++;
630 int SavingsExpected = Stores.size() - TotalInstsExpected;
631 if (SavingsExpected <= 0)
632 return false;
633
634 auto &MRI = MIB.getMF().getRegInfo();
635
636 // We have a series of consecutive stores. Factor out the common base
637 // pointer and rewrite the offsets.
638 Register NewBase = Stores[0].Ptr->getReg(0);
639 for (auto &SInfo : Stores) {
640 // Compute a new pointer with the new base ptr and adjusted offset.
641 MIB.setInstrAndDebugLoc(*SInfo.St);
642 auto NewOff =
643 MIB.buildConstant(LLT::integer(64), SInfo.Offset - BaseOffset);
644 auto NewPtr = MIB.buildPtrAdd(MRI.getType(SInfo.St->getPointerReg()),
645 NewBase, NewOff);
646 if (MIB.getObserver())
647 MIB.getObserver()->changingInstr(*SInfo.St);
648 SInfo.St->getOperand(1).setReg(NewPtr.getReg(0));
649 if (MIB.getObserver())
650 MIB.getObserver()->changedInstr(*SInfo.St);
651 }
652 LLVM_DEBUG(dbgs() << "Split a series of " << Stores.size()
653 << " stores into a base pointer and offsets.\n");
654 return true;
655}
656
657static bool optimizeConsecutiveMemOpAddressing(MachineFunction &MF,
658 CSEMIRBuilder &MIB) {
659 // This combine needs to run after all reassociations/folds on pointer
660 // addressing have been done, specifically those that combine two G_PTR_ADDs
661 // with constant offsets into a single G_PTR_ADD with a combined offset.
662 // The goal of this optimization is to undo that combine in the case where
663 // doing so has prevented the formation of pair stores due to illegal
664 // addressing modes of STP. The reason that we do it here is because
665 // it's much easier to undo the transformation of a series consecutive
666 // mem ops, than it is to detect when doing it would be a bad idea looking
667 // at a single G_PTR_ADD in the reassociation/ptradd_immed_chain combine.
668 //
669 // An example:
670 // G_STORE %11:_(<2 x s64>), %base:_(p0) :: (store (<2 x s64>), align 1)
671 // %off1:_(s64) = G_CONSTANT i64 4128
672 // %p1:_(p0) = G_PTR_ADD %0:_, %off1:_(s64)
673 // G_STORE %11:_(<2 x s64>), %p1:_(p0) :: (store (<2 x s64>), align 1)
674 // %off2:_(s64) = G_CONSTANT i64 4144
675 // %p2:_(p0) = G_PTR_ADD %0:_, %off2:_(s64)
676 // G_STORE %11:_(<2 x s64>), %p2:_(p0) :: (store (<2 x s64>), align 1)
677 // %off3:_(s64) = G_CONSTANT i64 4160
678 // %p3:_(p0) = G_PTR_ADD %0:_, %off3:_(s64)
679 // G_STORE %11:_(<2 x s64>), %17:_(p0) :: (store (<2 x s64>), align 1)
680 bool Changed = false;
681 auto &MRI = MF.getRegInfo();
682
684 .getCLOpts()
685 .postlegalizer_consecutive_memops)
686 return Changed;
687
689 // If we see a load, then we keep track of any values defined by it.
690 // In the following example, STP formation will fail anyway because
691 // the latter store is using a load result that appears after the
692 // the prior store. In this situation if we factor out the offset then
693 // we increase code size for no benefit.
694 // G_STORE %v1:_(s64), %base:_(p0) :: (store (s64))
695 // %v2:_(s64) = G_LOAD %ldptr:_(p0) :: (load (s64))
696 // G_STORE %v2:_(s64), %base:_(p0) :: (store (s64))
697 SmallVector<Register> LoadValsSinceLastStore;
698
699 auto storeIsValid = [&](StoreInfo &Last, StoreInfo New) {
700 // Check if this store is consecutive to the last one.
701 if (Last.Ptr->getBaseReg() != New.Ptr->getBaseReg() ||
702 (Last.Offset + static_cast<int64_t>(Last.StoredType.getSizeInBytes()) !=
703 New.Offset) ||
704 Last.StoredType != New.StoredType)
705 return false;
706
707 // Check if this store is using a load result that appears after the
708 // last store. If so, bail out.
709 if (any_of(LoadValsSinceLastStore, [&](Register LoadVal) {
710 return New.St->getValueReg() == LoadVal;
711 }))
712 return false;
713
714 // Check if the current offset would be too large for STP.
715 // If not, then STP formation should be able to handle it, so we don't
716 // need to do anything.
717 int64_t MaxLegalOffset;
718 switch (New.StoredType.getSizeInBits()) {
719 case 32:
720 MaxLegalOffset = 252;
721 break;
722 case 64:
723 MaxLegalOffset = 504;
724 break;
725 case 128:
726 MaxLegalOffset = 1008;
727 break;
728 default:
729 llvm_unreachable("Unexpected stored type size");
730 }
731 if (New.Offset < MaxLegalOffset)
732 return false;
733
734 // If factoring it out still wouldn't help then don't bother.
735 return New.Offset - Stores[0].Offset <= MaxLegalOffset;
736 };
737
738 auto resetState = [&]() {
739 Stores.clear();
740 LoadValsSinceLastStore.clear();
741 };
742
743 for (auto &MBB : MF) {
744 // We're looking inside a single BB at a time since the memset pattern
745 // should only be in a single block.
746 resetState();
747 for (auto &MI : MBB) {
748 // Skip for scalable vectors
749 if (auto *LdSt = dyn_cast<GLoadStore>(&MI);
750 LdSt && MRI.getType(LdSt->getOperand(0).getReg()).isScalableVector())
751 continue;
752
753 if (auto *St = dyn_cast<GStore>(&MI)) {
754 Register PtrBaseReg;
756 LLT StoredValTy = MRI.getType(St->getValueReg());
757 unsigned ValSize = StoredValTy.getSizeInBits();
758 if (ValSize < 32 || St->getMMO().getSizeInBits() != ValSize)
759 continue;
760
761 Register PtrReg = St->getPointerReg();
762 if (mi_match(
763 PtrReg, MRI,
764 m_OneNonDBGUse(m_GPtrAdd(m_Reg(PtrBaseReg), m_ICst(Offset))))) {
765 GPtrAdd *PtrAdd = cast<GPtrAdd>(MRI.getVRegDef(PtrReg));
766 StoreInfo New = {St, PtrAdd, Offset.getSExtValue(), StoredValTy};
767
768 if (Stores.empty()) {
769 Stores.push_back(New);
770 continue;
771 }
772
773 // Check if this store is a valid continuation of the sequence.
774 auto &Last = Stores.back();
775 if (storeIsValid(Last, New)) {
776 Stores.push_back(New);
777 LoadValsSinceLastStore.clear(); // Reset the load value tracking.
778 } else {
779 // The store isn't a valid to consider for the prior sequence,
780 // so try to optimize what we have so far and start a new sequence.
781 Changed |= tryOptimizeConsecStores(Stores, MIB);
782 resetState();
783 Stores.push_back(New);
784 }
785 }
786 } else if (auto *Ld = dyn_cast<GLoad>(&MI)) {
787 LoadValsSinceLastStore.push_back(Ld->getDstReg());
788 }
789 }
790 Changed |= tryOptimizeConsecStores(Stores, MIB);
791 resetState();
792 }
793
794 return Changed;
795}
796
797bool runCombiner(MachineFunction &MF, GISelCSEInfo *CSEInfo,
799 const AArch64PostLegalizerCombinerImplRuleConfig &RuleConfig,
800 bool EnableOpt, bool IsOptNone) {
801 if (MF.getProperties().hasFailedISel())
802 return false;
803 const Function &F = MF.getFunction();
804
806 const LegalizerInfo *LI = ST.getLegalizerInfo();
807
808 CombinerInfo CInfo(/*AllowIllegalOps=*/false, /*ShouldLegalizeIllegal=*/false,
809 /*LegalizerInfo=*/LI, EnableOpt, F.hasOptSize(),
810 F.hasMinSize());
811 // Disable fixed-point iteration to reduce compile-time
812 CInfo.MaxIterations = 1;
813 CInfo.ObserverLvl = CombinerInfo::ObserverLevel::SinglePass;
814 // Legalizer performs DCE, so a full DCE pass is unnecessary.
815 CInfo.EnableFullDCE = false;
816 AArch64PostLegalizerCombinerImpl Impl(MF, CInfo, *VT, CSEInfo, RuleConfig, ST,
817 MDT, LI);
818 bool Changed = Impl.combineMachineInstrs();
819
820 CSEMIRBuilder MIB(MF);
821 MIB.setCSEInfo(CSEInfo);
822 Changed |= optimizeConsecutiveMemOpAddressing(MF, MIB);
823 return Changed;
824}
825
826class AArch64PostLegalizerCombinerLegacy : public MachineFunctionPass {
827public:
828 static char ID;
829
830 AArch64PostLegalizerCombinerLegacy(bool IsOptNone = false);
831
832 StringRef getPassName() const override {
833 return "AArch64PostLegalizerCombiner";
834 }
835
836 bool runOnMachineFunction(MachineFunction &MF) override;
837 void getAnalysisUsage(AnalysisUsage &AU) const override;
838
839 MachineFunctionProperties getRequiredProperties() const override {
840 return MachineFunctionProperties().set(
841 MachineFunctionProperties::Property::Legalized);
842 }
843
844private:
845 bool IsOptNone;
846 AArch64PostLegalizerCombinerImplRuleConfig RuleConfig;
847};
848} // end anonymous namespace
849
850void AArch64PostLegalizerCombinerLegacy::getAnalysisUsage(
851 AnalysisUsage &AU) const {
852 AU.setPreservesCFG();
854 AU.addRequired<GISelValueTrackingAnalysisLegacy>();
855 AU.addPreserved<GISelValueTrackingAnalysisLegacy>();
856 if (!IsOptNone) {
857 AU.addRequired<MachineDominatorTreeWrapperPass>();
858 AU.addRequired<GISelCSEAnalysisWrapperPass>();
859 AU.addPreserved<GISelCSEAnalysisWrapperPass>();
860 }
862}
863
864AArch64PostLegalizerCombinerLegacy::AArch64PostLegalizerCombinerLegacy(
865 bool IsOptNone)
866 : MachineFunctionPass(ID), IsOptNone(IsOptNone) {
867 if (!RuleConfig.parseCommandLineOption())
868 reportFatalUsageError("Invalid rule identifier");
869}
870
871bool AArch64PostLegalizerCombinerLegacy::runOnMachineFunction(
872 MachineFunction &MF) {
873 if (MF.getProperties().hasFailedISel())
874 return false;
875
876 GISelValueTracking *VT =
877 &getAnalysis<GISelValueTrackingAnalysisLegacy>().get(MF);
878 MachineDominatorTree *MDT =
879 IsOptNone ? nullptr
880 : &getAnalysis<MachineDominatorTreeWrapperPass>().getDomTree();
881 GISelCSEAnalysisWrapper &Wrapper =
882 getAnalysis<GISelCSEAnalysisWrapperPass>().getCSEWrapper();
883 auto *CSEInfo =
885
886 bool EnableOpt = MF.getTarget().getOptLevel() != CodeGenOptLevel::None &&
887 !skipFunction(MF.getFunction());
888
889 return runCombiner(MF, CSEInfo, VT, MDT, RuleConfig, EnableOpt, IsOptNone);
890}
891
892char AArch64PostLegalizerCombinerLegacy::ID = 0;
893INITIALIZE_PASS_BEGIN(AArch64PostLegalizerCombinerLegacy, DEBUG_TYPE,
894 "Combine AArch64 MachineInstrs after legalization", false,
895 false)
897INITIALIZE_PASS_END(AArch64PostLegalizerCombinerLegacy, DEBUG_TYPE,
898 "Combine AArch64 MachineInstrs after legalization", false,
899 false)
900
903 : RuleConfig(
904 std::make_unique<AArch64PostLegalizerCombinerImplRuleConfig>()),
905 TM(TM) {
906 if (!RuleConfig->parseCommandLineOption())
907 reportFatalUsageError("invalid rule identifier");
908}
909
912
914
918 if (MF.getProperties().hasFailedISel())
919 return PreservedAnalyses::all();
920
921 const bool IsOptNone = TM->isGlobalISelOptNone();
922 bool EnableOpt =
924
927 IsOptNone ? nullptr : &MFAM.getResult<MachineDominatorTreeAnalysis>(MF);
928 GISelCSEInfo *CSEInfo = MFAM.getResult<GISelCSEAnalysis>(MF).get();
929
930 if (!runCombiner(MF, CSEInfo, VT, MDT, *RuleConfig, EnableOpt, IsOptNone))
931 return PreservedAnalyses::all();
932
937 return PA;
938}
939
940namespace llvm {
942 return new AArch64PostLegalizerCombinerLegacy(IsOptNone);
943}
944} // end namespace llvm
MachineInstrBuilder & UseMI
static bool isZeroExtended(SDValue N, SelectionDAG &DAG)
static bool isSignExtended(SDValue N, SelectionDAG &DAG)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
#define GET_GICOMBINER_CONSTRUCTOR_INITS
aarch64 promote const
amdgpu aa AMDGPU Address space based Alias Analysis Wrapper
MachineBasicBlock & MBB
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
Provides analysis for continuously CSEing during GISel passes.
This file implements a version of MachineIRBuilder which CSEs insts within a MachineBasicBlock.
This contains common combine transformations that may be used in a combine pass,or by the target else...
Option class for Targets to specify which operations are combined how and when.
This contains the base class for all Combiners generated by TableGen.
This contains common code to allow clients to notify changes to machine instr.
Provides analysis for querying information about KnownBits during GISel passes.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineIRBuilder class.
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
#define INITIALIZE_PASS_DEPENDENCY(depName)
Definition PassSupport.h:42
#define INITIALIZE_PASS_END(passName, arg, name, cfg, analysis)
Definition PassSupport.h:44
#define INITIALIZE_PASS_BEGIN(passName, arg, name, cfg, analysis)
Definition PassSupport.h:39
static StringRef getName(Value *V)
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
Value * RHS
Value * LHS
PreservedAnalyses run(MachineFunction &MF, MachineFunctionAnalysisManager &MFAM)
AArch64PostLegalizerCombinerPass(const AArch64TargetMachine *TM)
const AArch64Options & getCLOpts() const
Class for arbitrary precision integers.
Definition APInt.h:78
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
unsigned logBase2() const
Definition APInt.h:1781
APInt ashr(unsigned ShiftAmt) const
Arithmetic right-shift function.
Definition APInt.h:829
bool isNonNegative() const
Determine if this APInt Value is non-negative (>= 0)
Definition APInt.h:330
bool isPowerOf2() const
Check if this APInt's value is a power of two greater than zero.
Definition APInt.h:436
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
PassT::Result & getResult(IRUnitT &IR, ExtraArgTs... ExtraArgs)
Get the result of an analysis pass for a given IR unit.
Represent the analysis usage information of a pass.
AnalysisUsage & addRequired()
AnalysisUsage & addPreserved()
Add the specified Pass class to the set of analyses preserved by this pass.
LLVM_ABI void setPreservesCFG()
This function should be called by the pass, iff they do not:
Definition Pass.cpp:278
Represents analyses that only rely on functions' control flow.
Definition Analysis.h:73
Defines a builder that does CSE of MachineInstructions using GISelCSEInfo.
MachineInstrBuilder buildConstant(const DstOp &Res, const ConstantInt &Val) override
Build and insert Res = G_CONSTANT Val.
@ ICMP_SLT
signed less than
Definition InstrTypes.h:769
Combiner implementation.
Definition Combiner.h:33
FunctionPass class - This class is used to implement most global optimizations.
Definition Pass.h:314
The CSE Analysis object.
Definition CSEInfo.h:72
Abstract class that contains various methods for clients to notify about changes.
virtual void changingInstr(MachineInstr &MI)=0
This instruction is about to be mutated in some way.
virtual void changedInstr(MachineInstr &MI)=0
This instruction was mutated in some way.
To use KnownBitsInfo analysis in a pass, KnownBitsInfo &Info = getAnalysis<GISelValueTrackingInfoAnal...
bool maskedValueIsZero(Register Val, const APInt &Mask)
unsigned computeNumSignBits(Register R, const APInt &DemandedElts, unsigned Depth=0)
Represents a G_PTR_ADD.
Represents a G_STORE.
LLT changeElementCount(ElementCount EC) const
Return a vector or scalar with the same element type and the new element count.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr uint16_t getNumElements() const
Returns the number of elements in a vector LLT.
constexpr bool isVector() const
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr ElementCount getElementCount() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
static LLT integer(unsigned SizeInBits)
LLT changeElementSize(unsigned NewEltSize) const
If this type is a vector, return a vector with the same number of elements but the new element size.
Analysis pass which computes a MachineDominatorTree.
DominatorTree Class - Concrete subclass of DominatorTreeBase that is used to compute a normal dominat...
MachineFunctionPass - This class adapts the FunctionPass interface to allow convenient creation of pa...
void getAnalysisUsage(AnalysisUsage &AU) const override
getAnalysisUsage - Subclasses that override getAnalysisUsage must call this.
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
const MachineFunctionProperties & getProperties() const
Get the function properties.
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Helper class to build MachineInstr.
GISelChangeObserver * getObserver()
MachineInstrBuilder buildPtrAdd(const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_PTR_ADD Op0, Op1.
MachineFunction & getMF()
Getter for the function we currently build.
void setInstrAndDebugLoc(MachineInstr &MI)
Set the insertion point to before MI, and set the debug loc to MI's loc.
void setCSEInfo(GISelCSEInfo *Info)
Representation of each machine instruction.
unsigned getOpcode() const
Returns the opcode of this MachineInstr.
unsigned getNumOperands() const
Retuns the total number of operands.
const MachineOperand & getOperand(unsigned i) const
ArrayRef< int > getShuffleMask() const
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
use_instr_iterator use_instr_begin(Register RegNo) const
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
bool hasOneUse(Register RegNo) const
hasOneUse - Return true if there is exactly one instruction using the specified register.
iterator_range< use_instr_nodbg_iterator > use_nodbg_instructions(Register Reg) const
A set of analyses that are preserved following a run of a transformation pass.
Definition Analysis.h:112
static PreservedAnalyses all()
Construct a special preserved set that preserves all passes.
Definition Analysis.h:118
PreservedAnalyses & preserveSet()
Mark an analysis set as preserved.
Definition Analysis.h:151
PreservedAnalyses & preserve()
Mark an analysis as preserved.
Definition Analysis.h:132
Wrapper class representing virtual and physical registers.
Definition Register.h:20
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
virtual const TargetLowering * getTargetLowering() const
Changed
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
operand_type_match m_Reg()
UnaryOp_match< SrcTy, TargetOpcode::G_ZEXT > m_GZExt(const SrcTy &Src)
UnaryOp_match< SrcTy, TargetOpcode::G_SEXT > m_GSExt(const SrcTy &Src)
ConstantMatch< APInt > m_ICst(APInt &Cst)
BinaryOp_match< LHS, RHS, TargetOpcode::G_OR, true > m_GOr(const LHS &L, const RHS &R)
OneNonDBGUse_match< SubPat > m_OneNonDBGUse(const SubPat &SP)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, TargetOpcode::G_AND, true > m_GAnd(const LHS &L, const RHS &R)
SrcImmOp_match< SrcTy, AnyImmMatch, TargetOpcode::G_SEXT_INREG > m_GSExtInReg(const SrcTy &Src)
Matches a G_SEXT_INREG, binding its source and immediate width.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI std::optional< APInt > isConstantOrConstantSplatVector(Register Def, const MachineRegisterInfo &MRI)
Determines if Def defines a constant integer or a splat vector of constant integers.
Definition Utils.cpp:1517
@ Offset
Definition DWP.cpp:577
LLVM_ABI MachineInstr * getOpcodeDef(unsigned Opcode, Register Reg, const MachineRegisterInfo &MRI)
See if Reg is defined by an single def instruction that is Opcode.
Definition Utils.cpp:656
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Store
The extracted value is stored (ExtractElement only).
AnalysisManager< MachineFunction > MachineFunctionAnalysisManager
LLVM_ABI MachineInstr * getDefIgnoringCopies(Register Reg, const MachineRegisterInfo &MRI)
Find the def instruction for Reg, folding away any trivial copies.
Definition Utils.cpp:497
FunctionPass * createAArch64PostLegalizerCombinerLegacy(bool IsOptNone)
LLVM_ABI std::unique_ptr< CSEConfigBase > getStandardCSEConfigForOpt(CodeGenOptLevel Level)
Definition CSEInfo.cpp:85
unsigned M1(unsigned Val)
Definition VE.h:377
LLVM_ABI PreservedAnalyses getMachineFunctionPassPreservedAnalyses()
Returns the minimum set of Analyses that all machine function passes must preserve.
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
@ Other
Any other memory.
Definition ModRef.h:68
LLVM_ABI void getSelectionDAGFallbackAnalysisUsage(AnalysisUsage &AU)
Modify analysis usage so it preserves passes required for the SelectionDAG fallback.
Definition Utils.cpp:1137
@ Sub
Subtraction of integers.
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
Implement std::hash so that hash_code can be used in STL containers.
Definition BitVector.h:878
@ SinglePass
Enables Observer-based DCE and additional heuristics that retry combining defined and used instructio...