LLVM 24.0.0git
AArch64LegalizerInfo.cpp
Go to the documentation of this file.
1//===- AArch64LegalizerInfo.cpp ----------------------------------*- C++ -*-==//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the Machinelegalizer class for
10/// AArch64.
11/// \todo This should be generated by TableGen.
12//===----------------------------------------------------------------------===//
13
15#include "AArch64Subtarget.h"
16#include "llvm/ADT/STLExtras.h"
28#include "llvm/IR/Intrinsics.h"
29#include "llvm/IR/IntrinsicsAArch64.h"
30#include "llvm/IR/Type.h"
32#include <initializer_list>
33
34#define DEBUG_TYPE "aarch64-legalinfo"
35
36using namespace llvm;
37using namespace LegalizeActions;
38using namespace LegalizeMutations;
39using namespace LegalityPredicates;
40using namespace MIPatternMatch;
41
43 : ST(&ST) {
44 using namespace TargetOpcode;
45 const LLT p0 = LLT::pointer(0, 64);
46 const LLT s8 = LLT::scalar(8);
47 const LLT s16 = LLT::scalar(16);
48 const LLT s32 = LLT::scalar(32);
49 const LLT s64 = LLT::scalar(64);
50 const LLT s128 = LLT::scalar(128);
51 const LLT v16s8 = LLT::fixed_vector(16, 8);
52 const LLT v8s8 = LLT::fixed_vector(8, 8);
53 const LLT v4s8 = LLT::fixed_vector(4, 8);
54 const LLT v2s8 = LLT::fixed_vector(2, 8);
55 const LLT v8s16 = LLT::fixed_vector(8, 16);
56 const LLT v4s16 = LLT::fixed_vector(4, 16);
57 const LLT v2s16 = LLT::fixed_vector(2, 16);
58 const LLT v2s32 = LLT::fixed_vector(2, 32);
59 const LLT v4s32 = LLT::fixed_vector(4, 32);
60 const LLT v2s64 = LLT::fixed_vector(2, 64);
61 const LLT v2p0 = LLT::fixed_vector(2, p0);
62
63 const LLT nxv16s8 = LLT::scalable_vector(16, s8);
64 const LLT nxv8s16 = LLT::scalable_vector(8, s16);
65 const LLT nxv4s32 = LLT::scalable_vector(4, s32);
66 const LLT nxv2s64 = LLT::scalable_vector(2, s64);
67
68 const LLT bf16 = LLT::bfloat16();
69 const LLT v4bf16 = LLT::fixed_vector(4, bf16);
70 const LLT v8bf16 = LLT::fixed_vector(8, bf16);
71
72 const LLT f16 = LLT::float16();
73 const LLT v4f16 = LLT::fixed_vector(4, f16);
74 const LLT v8f16 = LLT::fixed_vector(8, f16);
75
76 const LLT f32 = LLT::float32();
77 const LLT v2f32 = LLT::fixed_vector(2, f32);
78 const LLT v4f32 = LLT::fixed_vector(4, f32);
79
80 const LLT f64 = LLT::float64();
81 const LLT v2f64 = LLT::fixed_vector(2, f64);
82
83 const LLT f128 = LLT::float128();
84
85 const LLT i8 = LLT::integer(8);
86 const LLT v8i8 = LLT::fixed_vector(8, i8);
87 const LLT v16i8 = LLT::fixed_vector(16, i8);
88
89 const LLT i16 = LLT::integer(16);
90 const LLT v8i16 = LLT::fixed_vector(8, i16);
91 const LLT v4i16 = LLT::fixed_vector(4, i16);
92
93 const LLT i32 = LLT::integer(32);
94 const LLT v2i32 = LLT::fixed_vector(2, i32);
95 const LLT v4i32 = LLT::fixed_vector(4, i32);
96
97 const LLT i64 = LLT::integer(64);
98 const LLT v2i64 = LLT::fixed_vector(2, i64);
99
100 const LLT i128 = LLT::integer(128);
101
102 const LLT nxv16i8 = LLT::scalable_vector(16, i8);
103 const LLT nxv8i16 = LLT::scalable_vector(8, i16);
104 const LLT nxv4i32 = LLT::scalable_vector(4, i32);
105 const LLT nxv2i64 = LLT::scalable_vector(2, i64);
106
107 std::initializer_list<LLT> PackedVectorAllTypeList = {/* Begin 128bit types */
108 v16s8, v8s16, v4s32,
109 v2s64, v2p0,
110 /* End 128bit types */
111 /* Begin 64bit types */
112 v8s8, v4s16, v2s32};
113 std::initializer_list<LLT> ScalarAndPtrTypesList = {s8, s16, s32, s64, p0};
114 SmallVector<LLT, 8> PackedVectorAllTypesVec(PackedVectorAllTypeList);
115 SmallVector<LLT, 8> ScalarAndPtrTypesVec(ScalarAndPtrTypesList);
116
117 const TargetMachine &TM = ST.getTargetLowering()->getTargetMachine();
118
119 // FIXME: support subtargets which have neon/fp-armv8 disabled.
120 if (!ST.hasNEON() || !ST.hasFPARMv8())
121 return;
122
123 // Some instructions only support s16 if the subtarget has full 16-bit FP
124 // support.
125 const bool HasFP16 = ST.hasFullFP16();
126 const bool HasCSSC = ST.hasCSSC();
127 const bool HasRCPC3 = ST.hasRCPC3();
128 const bool HasSVE = ST.hasSVE();
129
131 {G_IMPLICIT_DEF, G_FREEZE, G_CONSTANT_FOLD_BARRIER})
132 .legalFor({p0, s8, s16, s32, s64, s128})
133 .legalFor({v2s8, v4s8, v8s8, v16s8, v2s16, v4s16, v8s16, v2s32, v4s32,
134 v2s64, v2p0})
135 .widenScalarToNextPow2(0)
136 .clampScalar(0, s8, s64)
139 .clampNumElements(0, v8s8, v16s8)
140 .clampNumElements(0, v4s16, v8s16)
141 .clampNumElements(0, v2s32, v4s32)
142 .clampMaxNumElements(0, s64, 2)
143 .clampMaxNumElements(0, p0, 2)
145
147 .legalFor({p0, s16, s32, s64})
148 .legalFor(PackedVectorAllTypeList)
152 .clampScalar(0, s16, s64)
153 .clampNumElements(0, v8s8, v16s8)
154 .clampNumElements(0, v4s16, v8s16)
155 .clampNumElements(0, v2s32, v4s32)
156 .clampMaxNumElements(0, s64, 2)
157 .clampMaxNumElements(0, p0, 2)
159
161 .legalIf(all(typeInSet(0, {s32, s64, p0}), typeInSet(1, {s8, s16, s32}),
162 smallerThan(1, 0)))
163 .widenScalarToNextPow2(0)
164 .clampScalar(0, s32, s64)
166 .minScalar(1, s8)
167 .maxScalarIf(typeInSet(0, {s32}), 1, s16)
168 .maxScalarIf(typeInSet(0, {s64, p0}), 1, s32);
169
171 .legalIf(all(typeInSet(0, {s16, s32, s64, p0}),
172 typeInSet(1, {s32, s64, s128, p0}), smallerThan(0, 1)))
173 .widenScalarToNextPow2(1)
174 .clampScalar(1, s32, s128)
176 .minScalar(0, s16)
177 .maxScalarIf(typeInSet(1, {s32}), 0, s16)
178 .maxScalarIf(typeInSet(1, {s64, p0}), 0, s32)
179 .maxScalarIf(typeInSet(1, {s128}), 0, s64);
180
181 getActionDefinitionsBuilder({G_ADD, G_SUB, G_AND, G_OR, G_XOR})
182 .legalFor({i32, i64, v8i8, v16i8, v4i16, v8i16, v2i32, v4i32, v2i64})
183 .legalFor(HasSVE, {nxv16i8, nxv8i16, nxv4i32, nxv2i64})
184 .widenScalarToNextPow2(0)
185 .clampScalar(0, s32, s64)
186 .clampMaxNumElements(0, s8, 16)
187 .clampMaxNumElements(0, s16, 8)
188 .clampNumElements(0, v2s32, v4s32)
189 .clampNumElements(0, v2s64, v2s64)
191 [=](const LegalityQuery &Query) {
192 return Query.Types[0].getNumElements() <= 2;
193 },
194 0, s32)
195 .minScalarOrEltIf(
196 [=](const LegalityQuery &Query) {
197 return Query.Types[0].getNumElements() <= 4;
198 },
199 0, s16)
200 .minScalarOrEltIf(
201 [=](const LegalityQuery &Query) {
202 return Query.Types[0].getNumElements() <= 16;
203 },
204 0, s8)
205 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
207
209 .legalFor({i32, i64, v8i8, v16i8, v4i16, v8i16, v2i32, v4i32, v2i64})
210 .widenScalarToNextPow2(0)
211 .clampScalar(0, s32, s64)
212 .clampMaxNumElements(0, s8, 16)
213 .clampMaxNumElements(0, s16, 8)
214 .clampNumElements(0, v2s32, v4s32)
215 .clampNumElements(0, v2s64, v2s64)
217 [=](const LegalityQuery &Query) {
218 return Query.Types[0].getNumElements() <= 2;
219 },
220 0, s32)
221 .minScalarOrEltIf(
222 [=](const LegalityQuery &Query) {
223 return Query.Types[0].getNumElements() <= 4;
224 },
225 0, s16)
226 .minScalarOrEltIf(
227 [=](const LegalityQuery &Query) {
228 return Query.Types[0].getNumElements() <= 16;
229 },
230 0, s8)
231 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
233
234 getActionDefinitionsBuilder({G_SHL, G_ASHR, G_LSHR})
235 .customIf([=](const LegalityQuery &Query) {
236 const auto &SrcTy = Query.Types[0];
237 const auto &AmtTy = Query.Types[1];
238 return !SrcTy.isVector() && SrcTy.getSizeInBits() == 32 &&
239 AmtTy.getSizeInBits() == 32;
240 })
241 .legalFor({
242 {i32, i32},
243 {i32, i64},
244 {i64, i64},
245 {v8i8, v8i8},
246 {v16i8, v16i8},
247 {v4i16, v4i16},
248 {v8i16, v8i16},
249 {v2i32, v2i32},
250 {v4i32, v4i32},
251 {v2i64, v2i64},
252 })
253 .widenScalarToNextPow2(1)
255 .clampScalar(1, s32, s64)
256 .clampScalar(0, s32, s64)
257 .clampNumElements(0, v8s8, v16s8)
258 .clampNumElements(0, v4s16, v8s16)
259 .clampNumElements(0, v2s32, v4s32)
260 .clampNumElements(0, v2s64, v2s64)
262 .minScalarSameAs(1, 0)
266
268 .legalFor({{p0, i64}, {v2p0, v2i64}})
269 .clampScalarOrElt(1, s64, s64)
270 .clampNumElements(0, v2p0, v2p0);
271
272 getActionDefinitionsBuilder(G_PTRMASK).legalFor({{p0, s64}});
273
274 getActionDefinitionsBuilder({G_SDIV, G_UDIV})
275 .legalFor({i32, i64})
276 .libcallFor({i128})
277 .clampScalar(0, s32, s64)
279 .scalarize(0);
280
281 getActionDefinitionsBuilder({G_SREM, G_UREM, G_SDIVREM, G_UDIVREM})
282 .lowerFor({i8, i16, i32, i64, v2i32, v4i32, v2i64})
283 .libcallFor({i128})
285 .minScalarOrElt(0, s32)
286 .clampNumElements(0, v2s32, v4s32)
287 .clampNumElements(0, v2s64, v2s64)
288 .scalarize(0);
289
290 getActionDefinitionsBuilder({G_SMULO, G_UMULO})
291 .widenScalarToNextPow2(0, /*Min = */ 32)
292 .clampScalar(0, s32, s64)
293 .lower();
294
295 getActionDefinitionsBuilder({G_SMULH, G_UMULH})
296 .legalFor({i64, v16i8, v8i16, v4i32})
297 .lower();
298
300 {G_SMULFIX, G_UMULFIX, G_SMULFIXSAT, G_UMULFIXSAT})
301 .lower();
302
303 getActionDefinitionsBuilder({G_SMIN, G_SMAX, G_UMIN, G_UMAX})
304 .legalFor({v8i8, v16i8, v4i16, v8i16, v2i32, v4i32})
305 .legalFor(HasCSSC, {i32, i64})
306 .minScalar(HasCSSC, 0, s32)
307 .clampNumElements(0, v8s8, v16s8)
308 .clampNumElements(0, v4s16, v8s16)
309 .clampNumElements(0, v2s32, v4s32)
310 .lower();
311
312 // FIXME: Legal vector types are only legal with NEON.
314 .legalFor(HasCSSC, {i32, i64})
315 .legalFor({v16i8, v8i16, v4i32, v2i64, v2p0, v8i8, v4i16, v2i32})
316 .customIf([=](const LegalityQuery &Q) {
317 // TODO: Fix suboptimal codegen for 128+ bit types.
318 LLT SrcTy = Q.Types[0];
319 return SrcTy.isScalar() && SrcTy.getSizeInBits() < 128;
320 })
321 .widenScalarIf(
322 [=](const LegalityQuery &Query) { return Query.Types[0] == v4s8; },
323 [=](const LegalityQuery &Query) { return std::make_pair(0, v4i16); })
324 .widenScalarIf(
325 [=](const LegalityQuery &Query) { return Query.Types[0] == v2s16; },
326 [=](const LegalityQuery &Query) { return std::make_pair(0, v2i32); })
327 .clampNumElements(0, v8s8, v16s8)
328 .clampNumElements(0, v4s16, v8s16)
329 .clampNumElements(0, v2s32, v4s32)
330 .clampNumElements(0, v2s64, v2s64)
332 .lower();
333
335 {G_ABDS, G_ABDU, G_UAVGFLOOR, G_UAVGCEIL, G_SAVGFLOOR, G_SAVGCEIL})
336 .legalFor({v8i8, v16i8, v4i16, v8i16, v2i32, v4i32})
337 .lower();
338
340 {G_SADDE, G_SSUBE, G_UADDE, G_USUBE, G_SADDO, G_SSUBO, G_UADDO, G_USUBO})
341 .legalFor({{i32, i32}, {i64, i32}})
342 .clampScalar(0, s32, s64)
343 .clampScalar(1, s32, s64)
345 .lower();
346
347 getActionDefinitionsBuilder({G_FSHL, G_FSHR})
348 .customFor({{i32, i32}, {i32, i64}, {i64, i64}})
349 .lower();
350
352 .legalFor({{i32, i64}, {i64, i64}})
353 .customIf([=](const LegalityQuery &Q) {
354 return Q.Types[0].isScalar() && Q.Types[1].getScalarSizeInBits() < 64;
355 })
356 .lower();
358
359 getActionDefinitionsBuilder({G_SBFX, G_UBFX})
360 .customFor({{s32, s32}, {s64, s64}});
361
362 auto always = [=](const LegalityQuery &Q) { return true; };
364 .legalFor(HasCSSC, {{i32, i32}, {i64, i64}})
365 .legalFor({{v8i8, v8i8}, {v16i8, v16i8}})
366 .customFor(!HasCSSC, {{s32, s32}, {s64, s64}})
367 .customFor({{s128, s128},
368 {v4s16, v4s16},
369 {v8s16, v8s16},
370 {v2s32, v2s32},
371 {v4s32, v4s32},
372 {v2s64, v2s64}})
373 .clampScalar(0, s32, s128)
376 .minScalarEltSameAsIf(always, 1, 0)
377 .maxScalarEltSameAsIf(always, 1, 0)
378 .clampNumElements(0, v8s8, v16s8)
379 .clampNumElements(0, v4s16, v8s16)
380 .clampNumElements(0, v2s32, v4s32)
381 .clampNumElements(0, v2s64, v2s64)
384
385 getActionDefinitionsBuilder({G_CTLZ, G_CTLS})
386 .legalFor({{i32, i32},
387 {i64, i64},
388 {v8i8, v8i8},
389 {v16i8, v16i8},
390 {v4i16, v4i16},
391 {v8i16, v8i16},
392 {v2i32, v2i32},
393 {v4i32, v4i32}})
394 .widenScalarToNextPow2(1, /*Min=*/32)
395 .clampScalar(1, s32, s64)
397 .clampNumElements(0, v8s8, v16s8)
398 .clampNumElements(0, v4s16, v8s16)
399 .clampNumElements(0, v2s32, v4s32)
402 .scalarSameSizeAs(0, 1);
403
404 getActionDefinitionsBuilder(G_INSERT_SUBVECTOR).lower();
405
406 getActionDefinitionsBuilder(G_CTLZ_ZERO_POISON).lower();
407
409 .lowerIf(isVector(0))
410 .widenScalarToNextPow2(1, /*Min=*/32)
411 .clampScalar(1, s32, s64)
412 .scalarSameSizeAs(0, 1)
413 .legalFor(HasCSSC, {s32, s64})
414 .customFor(!HasCSSC, {s32, s64});
415
416 getActionDefinitionsBuilder(G_CTTZ_ZERO_POISON).lower();
417
418 getActionDefinitionsBuilder(G_BITREVERSE)
419 .legalFor({i32, i64, v8i8, v16i8})
420 .widenScalarToNextPow2(0, /*Min = */ 32)
422 .clampScalar(0, s32, s64)
423 .clampNumElements(0, v8s8, v16s8)
424 .clampNumElements(0, v4s16, v8s16)
425 .clampNumElements(0, v2s32, v4s32)
426 .clampNumElements(0, v2s64, v2s64)
429 .lower();
430
431 getActionDefinitionsBuilder(G_CLMUL).legalFor({v8i8, v16i8});
432
434 .legalFor({i32, i64, v4i16, v8i16, v2i32, v4i32, v2i64})
436 .clampScalar(0, s32, s64)
437 .clampNumElements(0, v4s16, v8s16)
438 .clampNumElements(0, v2s32, v4s32)
439 .clampNumElements(0, v2s64, v2s64)
441
442 getActionDefinitionsBuilder({G_UADDSAT, G_SADDSAT, G_USUBSAT, G_SSUBSAT})
443 .legalFor({v8i8, v16i8, v4i16, v8i16, v2i32, v4i32, v2i64})
444 .legalFor(HasSVE, {nxv16i8, nxv8i16, nxv4i32, nxv2i64})
445 .clampNumElements(0, v8s8, v16s8)
446 .clampNumElements(0, v4s16, v8s16)
447 .clampNumElements(0, v2s32, v4s32)
448 .clampMaxNumElements(0, s64, 2)
451 .lower();
452
454 {G_FADD, G_FSUB, G_FMUL, G_FDIV, G_FMA, G_FSQRT, G_FMAXNUM, G_FMINNUM,
455 G_FMAXIMUM, G_FMINIMUM, G_FCEIL, G_FFLOOR, G_FRINT, G_FNEARBYINT,
456 G_INTRINSIC_TRUNC, G_INTRINSIC_ROUND, G_INTRINSIC_ROUNDEVEN})
457 .legalFor({f32, f64, v2f32, v4f32, v2f64})
458 .legalFor(HasFP16, {f16, v4f16, v8f16})
459 .libcallFor({f128})
460 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
462 [=](const LegalityQuery &Q) {
463 return (!HasFP16 && Q.Types[0].getScalarType().isFloat16()) ||
464 Q.Types[0].getScalarType().isBFloat16();
465 },
466 changeElementTo(0, f32))
467 .clampNumElements(0, v4s16, v8s16)
468 .clampNumElements(0, v2s32, v4s32)
469 .clampNumElements(0, v2s64, v2s64)
471
472 getActionDefinitionsBuilder({G_FABS, G_FNEG})
473 .legalFor({f32, f64, v2f32, v4f32, v2f64})
474 .legalFor(HasFP16, {f16, bf16, v4f16, v4bf16, v8f16, v8bf16})
475 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
477 .clampNumElements(0, v4s16, v8s16)
478 .clampNumElements(0, v2s32, v4s32)
479 .clampNumElements(0, v2s64, v2s64)
481 .lowerFor({f16, bf16, v4f16, v4bf16, v8f16, v8bf16});
482
483 getActionDefinitionsBuilder({G_FREM, G_FCOS, G_FSIN, G_FPOW, G_FLOG, G_FLOG2,
484 G_FLOG10, G_FTAN, G_FEXP, G_FEXP2, G_FEXP10,
485 G_FACOS, G_FASIN, G_FATAN, G_FATAN2, G_FCOSH,
486 G_FSINH, G_FTANH, G_FMODF})
487 .libcallFor({f32, f64, f128})
488 .widenScalarFor({f16, bf16}, changeElementTo(0, f32))
489 .scalarize(0);
490 getActionDefinitionsBuilder({G_FPOWI, G_FLDEXP})
491 .libcallFor({{f32, i32}, {f64, i32}, {f128, i32}})
492 .widenScalarFor({f16, bf16}, changeElementTo(0, f32))
493 .scalarize(0);
494
495 getActionDefinitionsBuilder({G_LROUND, G_INTRINSIC_LRINT})
496 .legalFor({{i32, f32}, {i32, f64}, {i64, f32}, {i64, f64}})
497 .legalFor(HasFP16, {{i32, f16}, {i64, f16}})
498 .minScalar(1, s32)
499 .libcallFor({{s64, s128}})
500 .lower();
501 getActionDefinitionsBuilder({G_LLROUND, G_INTRINSIC_LLRINT})
502 .legalFor({{i64, f32}, {i64, f64}})
503 .legalFor(HasFP16, {{i64, f16}})
504 .minScalar(0, s64)
505 .minScalar(1, s32)
506 .libcallFor({{s64, s128}})
507 .lower();
508
509 // TODO: Custom legalization for mismatched types.
510 getActionDefinitionsBuilder(G_FCOPYSIGN)
512 [](const LegalityQuery &Query) { return Query.Types[0].isScalar(); },
513 [=](const LegalityQuery &Query) {
514 const LLT Ty = Query.Types[0];
515 return std::pair(0, LLT::fixed_vector(Ty == s16 ? 4 : 2, Ty));
516 })
517 .lower();
518
520
521 for (unsigned Op : {G_SEXTLOAD, G_ZEXTLOAD}) {
522 auto &Actions = getActionDefinitionsBuilder(Op);
523
524 if (Op == G_SEXTLOAD)
526
527 // Atomics have zero extending behavior.
528 Actions
529 .legalForTypesWithMemDesc({{s32, p0, s8, 8},
530 {s32, p0, s16, 8},
531 {s32, p0, s32, 8},
532 {s64, p0, s8, 2},
533 {s64, p0, s16, 2},
534 {s64, p0, s32, 4},
535 {s64, p0, s64, 8},
536 {p0, p0, s64, 8},
537 {v2s32, p0, s64, 8}})
538 .widenScalarToNextPow2(0)
539 .clampScalar(0, s32, s64)
540 // TODO: We could support sum-of-pow2's but the lowering code doesn't know
541 // how to do that yet.
542 .unsupportedIfMemSizeNotPow2()
543 // Lower anything left over into G_*EXT and G_LOAD
544 .lower();
545 }
546
547 auto IsPtrVecPred = [=](const LegalityQuery &Query) {
548 const LLT &ValTy = Query.Types[0];
549 return ValTy.isPointerVector() && ValTy.getAddressSpace() == 0;
550 };
551
553 .customIf([=](const LegalityQuery &Query) {
554 return HasRCPC3 && Query.Types[0] == s128 &&
555 Query.MMODescrs[0].Ordering == AtomicOrdering::Acquire;
556 })
557 .customIf([=](const LegalityQuery &Query) {
558 return Query.Types[0] == s128 &&
559 Query.MMODescrs[0].Ordering != AtomicOrdering::NotAtomic;
560 })
561 .legalForTypesWithMemDesc({{s8, p0, s8, 8},
562 {s16, p0, s16, 8},
563 {s32, p0, s32, 8},
564 {s64, p0, s64, 8},
565 {p0, p0, s64, 8},
566 {s128, p0, s128, 8},
567 {v8s8, p0, s64, 8},
568 {v16s8, p0, s128, 8},
569 {v4s16, p0, s64, 8},
570 {v8s16, p0, s128, 8},
571 {v2s32, p0, s64, 8},
572 {v4s32, p0, s128, 8},
573 {v2s64, p0, s128, 8}})
574 // These extends are also legal
575 .legalForTypesWithMemDesc(
576 {{s32, p0, s8, 8}, {s32, p0, s16, 8}, {s64, p0, s32, 8}})
577 .legalForTypesWithMemDesc({
578 // SVE vscale x 128 bit base sizes
579 {nxv16s8, p0, nxv16s8, 8},
580 {nxv8s16, p0, nxv8s16, 8},
581 {nxv4s32, p0, nxv4s32, 8},
582 {nxv2s64, p0, nxv2s64, 8},
583 })
584 .widenScalarToNextPow2(0, /* MinSize = */ 8)
585 .clampMaxNumElements(0, s8, 16)
586 .clampMaxNumElements(0, s16, 8)
587 .clampMaxNumElements(0, s32, 4)
588 .clampMaxNumElements(0, s64, 2)
589 .clampMaxNumElements(0, p0, 2)
591 .clampScalar(0, s8, s64)
593 [=](const LegalityQuery &Query) {
594 // Clamp extending load results to 32-bits.
595 return Query.Types[0].isScalar() &&
596 Query.Types[0] != Query.MMODescrs[0].MemoryTy &&
597 Query.Types[0].getSizeInBits() > 32;
598 },
599 changeTo(0, i32))
600 // TODO: Use BITCAST for v2i8, v2i16 after G_TRUNC gets sorted out
601 .bitcastIf(typeInSet(0, {v4s8}),
602 [=](const LegalityQuery &Query) {
603 const LLT VecTy = Query.Types[0];
604 return std::pair(0, LLT::integer(VecTy.getSizeInBits()));
605 })
606 .customIf(IsPtrVecPred)
607 .scalarizeIf(typeInSet(0, {v2s16, v2s8}), 0)
608 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0);
609
611 .customIf([=](const LegalityQuery &Query) {
612 return HasRCPC3 && Query.Types[0] == s128 &&
613 Query.MMODescrs[0].Ordering == AtomicOrdering::Release;
614 })
615 .customIf([=](const LegalityQuery &Query) {
616 return Query.Types[0] == s128 &&
617 Query.MMODescrs[0].Ordering != AtomicOrdering::NotAtomic;
618 })
619 .widenScalarIf(
620 all(scalarNarrowerThan(0, 32),
622 changeElementSizeTo(0, s32))
624 {{s8, p0, s8, 8}, {s16, p0, s8, 8}, // truncstorei8 from s16
625 {s32, p0, s8, 8}, // truncstorei8 from s32
626 {s64, p0, s8, 8}, // truncstorei8 from s64
627 {s16, p0, s16, 8}, {s32, p0, s16, 8}, // truncstorei16 from s32
628 {s64, p0, s16, 8}, // truncstorei16 from s64
629 {s32, p0, s8, 8}, {s32, p0, s16, 8}, {s32, p0, s32, 8},
630 {s64, p0, s64, 8}, {s64, p0, s32, 8}, // truncstorei32 from s64
631 {p0, p0, s64, 8}, {s128, p0, s128, 8}, {v16s8, p0, s128, 8},
632 {v8s8, p0, s64, 8}, {v4s16, p0, s64, 8}, {v8s16, p0, s128, 8},
633 {v2s32, p0, s64, 8}, {v4s32, p0, s128, 8}, {v2s64, p0, s128, 8}})
634 .legalForTypesWithMemDesc({
635 // SVE vscale x 128 bit base sizes
636 // TODO: Add nxv2p0. Consider bitcastIf.
637 // See #92130
638 // https://github.com/llvm/llvm-project/pull/92130#discussion_r1616888461
639 {nxv16s8, p0, nxv16s8, 8},
640 {nxv8s16, p0, nxv8s16, 8},
641 {nxv4s32, p0, nxv4s32, 8},
642 {nxv2s64, p0, nxv2s64, 8},
643 })
644 .clampScalar(0, s8, s64)
645 .minScalarOrElt(0, s8)
646 .lowerIf([=](const LegalityQuery &Query) {
647 return Query.Types[0].isScalar() &&
648 Query.Types[0] != Query.MMODescrs[0].MemoryTy;
649 })
650 // Maximum: sN * k = 128
651 .clampMaxNumElements(0, s8, 16)
652 .clampMaxNumElements(0, s16, 8)
653 .clampMaxNumElements(0, s32, 4)
654 .clampMaxNumElements(0, s64, 2)
655 .clampMaxNumElements(0, p0, 2)
657 // TODO: Use BITCAST for v2i8, v2i16 after G_TRUNC gets sorted out
658 .bitcastIf(all(typeInSet(0, {v4s8}),
659 LegalityPredicate([=](const LegalityQuery &Query) {
660 return Query.Types[0].getSizeInBits() ==
661 Query.MMODescrs[0].MemoryTy.getSizeInBits();
662 })),
663 [=](const LegalityQuery &Query) {
664 const LLT VecTy = Query.Types[0];
665 return std::pair(0, LLT::integer(VecTy.getSizeInBits()));
666 })
667 .customIf(IsPtrVecPred)
668 .scalarizeIf(typeInSet(0, {v2s16, v2s8}), 0)
669 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
670 .lower();
671
672 getActionDefinitionsBuilder(G_INDEXED_STORE)
673 // Idx 0 == Ptr, Idx 1 == Val
674 // TODO: we can implement legalizations but as of now these are
675 // generated in a very specific way.
677 {p0, s8, s8, 8},
678 {p0, s16, s16, 8},
679 {p0, s32, s8, 8},
680 {p0, s32, s16, 8},
681 {p0, s32, s32, 8},
682 {p0, s64, s64, 8},
683 {p0, p0, p0, 8},
684 {p0, v8s8, v8s8, 8},
685 {p0, v16s8, v16s8, 8},
686 {p0, v4s16, v4s16, 8},
687 {p0, v8s16, v8s16, 8},
688 {p0, v2s32, v2s32, 8},
689 {p0, v4s32, v4s32, 8},
690 {p0, v2s64, v2s64, 8},
691 {p0, v2p0, v2p0, 8},
692 {p0, s128, s128, 8},
693 })
694 .unsupported();
695
696 auto IndexedLoadBasicPred = [=](const LegalityQuery &Query) {
697 LLT LdTy = Query.Types[0];
698 LLT PtrTy = Query.Types[1];
699 if (!llvm::is_contained(PackedVectorAllTypesVec, LdTy) &&
700 !llvm::is_contained(ScalarAndPtrTypesVec, LdTy) && LdTy != s128)
701 return false;
702 if (PtrTy != p0)
703 return false;
704 return true;
705 };
706 getActionDefinitionsBuilder(G_INDEXED_LOAD)
709 .legalIf(IndexedLoadBasicPred)
710 .unsupported();
711 getActionDefinitionsBuilder({G_INDEXED_SEXTLOAD, G_INDEXED_ZEXTLOAD})
712 .unsupportedIf(
714 .legalIf(all(typeInSet(0, {s16, s32, s64}),
715 LegalityPredicate([=](const LegalityQuery &Q) {
716 LLT LdTy = Q.Types[0];
717 LLT PtrTy = Q.Types[1];
718 LLT MemTy = Q.MMODescrs[0].MemoryTy;
719 if (PtrTy != p0)
720 return false;
721 if (LdTy == s16)
722 return MemTy == s8;
723 if (LdTy == s32)
724 return MemTy == s8 || MemTy == s16;
725 if (LdTy == s64)
726 return MemTy == s8 || MemTy == s16 || MemTy == s32;
727 return false;
728 })))
729 .unsupported();
730
731 // Constants
733 .legalFor({p0, s8, s16, s32, s64})
734 .widenScalarToNextPow2(0)
735 .clampScalar(0, s8, s64);
736 getActionDefinitionsBuilder(G_FCONSTANT)
737 .legalFor({s16, s32, s64, s128});
738
739 // FIXME: fix moreElementsToNextPow2
741 .legalFor({{i32, i32}, {i32, i64}, {i32, p0}})
743 .minScalarOrElt(1, s8)
744 .clampScalar(1, s32, s64)
745 .clampScalar(0, s32, s32)
748 [=](const LegalityQuery &Query) {
749 const LLT &Ty = Query.Types[0];
750 const LLT &SrcTy = Query.Types[1];
751 return Ty.isVector() && !SrcTy.isPointerVector() &&
752 Ty.getElementType() != SrcTy.getElementType();
753 },
754 0, 1)
755 .minScalarOrEltIf(
756 [=](const LegalityQuery &Query) { return Query.Types[1] == v2s16; },
757 1, s32)
758 .minScalarOrEltIf(
759 [=](const LegalityQuery &Query) {
760 return Query.Types[1].isPointerVector();
761 },
762 0, s64)
764 .clampNumElements(1, v8s8, v16s8)
765 .clampNumElements(1, v4s16, v8s16)
766 .clampNumElements(1, v2s32, v4s32)
767 .clampNumElements(1, v2s64, v2s64)
768 .clampNumElements(1, v2p0, v2p0)
769 .customIf(isVector(0));
770
772 .legalFor({{i32, f32},
773 {i32, f64},
774 {v4i32, v4f32},
775 {v2i32, v2f32},
776 {v2i64, v2f64}})
777 .legalFor(HasFP16, {{i32, f16}, {v4i16, v4f16}, {v8i16, v8f16}})
779 .clampScalar(0, s32, s32)
781 [=](const LegalityQuery &Q) {
782 return (!HasFP16 && Q.Types[1].getScalarType().isFloat16()) ||
783 Q.Types[1].getScalarType().isBFloat16();
784 },
785 changeElementTo(1, f32))
786 .scalarizeIf(scalarOrEltWiderThan(1, 64), 1)
788 [=](const LegalityQuery &Query) {
789 const LLT &Ty = Query.Types[0];
790 const LLT &SrcTy = Query.Types[1];
791 return Ty.isVector() && !SrcTy.isPointerVector() &&
792 Ty.getElementType() != SrcTy.getElementType();
793 },
794 0, 1)
795 .clampNumElements(1, v4s16, v8s16)
796 .clampNumElements(1, v2s32, v4s32)
797 .clampMaxNumElements(1, s64, 2)
799 .libcallFor({{s32, s128}});
800
801 // Extensions
802 auto ExtLegalFunc = [=](const LegalityQuery &Query) {
803 unsigned DstSize = Query.Types[0].getSizeInBits();
804
805 // Handle legal vectors using legalFor
806 if (Query.Types[0].isVector())
807 return false;
808
809 if (DstSize < 8 || DstSize >= 128 || !isPowerOf2_32(DstSize))
810 return false; // Extending to a scalar s128 needs narrowing.
811
812 const LLT &SrcTy = Query.Types[1];
813
814 // Make sure we fit in a register otherwise. Don't bother checking that
815 // the source type is below 128 bits. We shouldn't be allowing anything
816 // through which is wider than the destination in the first place.
817 unsigned SrcSize = SrcTy.getSizeInBits();
818 if (SrcSize < 8 || !isPowerOf2_32(SrcSize))
819 return false;
820
821 return true;
822 };
823 getActionDefinitionsBuilder({G_ZEXT, G_SEXT, G_ANYEXT})
824 .legalIf(ExtLegalFunc)
825 .legalFor({{v8s16, v8s8}, {v4s32, v4s16}, {v2s64, v2s32}})
826 .clampScalar(0, s64, s64) // Just for s128, others are handled above.
828 .clampMaxNumElements(1, s8, 8)
829 .clampMaxNumElements(1, s16, 4)
830 .clampMaxNumElements(1, s32, 2)
831 // Tries to convert a large EXTEND into two smaller EXTENDs
832 .lowerIf([=](const LegalityQuery &Query) {
833 return (Query.Types[0].getScalarSizeInBits() >
834 Query.Types[1].getScalarSizeInBits() * 2) &&
835 Query.Types[0].isVector() &&
836 (Query.Types[1].getScalarSizeInBits() == 8 ||
837 Query.Types[1].getScalarSizeInBits() == 16);
838 })
839 .clampMinNumElements(1, s8, 8)
840 .clampMinNumElements(1, s16, 4)
842
844 .legalFor({{v8s8, v8s16}, {v4s16, v4s32}, {v2s32, v2s64}})
846 .clampMaxNumElements(0, s8, 8)
847 .clampMaxNumElements(0, s16, 4)
848 .clampMaxNumElements(0, s32, 2)
850 [=](const LegalityQuery &Query) { return Query.Types[0].isVector(); },
851 0, s8)
852 .lowerIf([=](const LegalityQuery &Query) {
853 LLT DstTy = Query.Types[0];
854 LLT SrcTy = Query.Types[1];
855 return DstTy.isVector() && SrcTy.getSizeInBits() > 128 &&
856 DstTy.getScalarSizeInBits() * 2 <= SrcTy.getScalarSizeInBits();
857 })
858 .clampMinNumElements(0, s8, 8)
859 .clampMinNumElements(0, s16, 4)
860 .alwaysLegal();
861
862 getActionDefinitionsBuilder({G_TRUNC_SSAT_S, G_TRUNC_SSAT_U, G_TRUNC_USAT_U})
863 .legalFor({{v8i8, v8i16}, {v4i16, v4i32}, {v2i32, v2i64}})
864 .clampNumElements(0, v8s8, v8s8)
865 .clampNumElements(0, v4s16, v4s16)
866 .clampNumElements(0, v2s32, v2s32)
867 .lower();
868
869 getActionDefinitionsBuilder(G_SEXT_INREG)
870 .legalFor({i32, i64, v8i8, v16i8, v4i16, v8i16, v2i32, v4i32, v2i64})
871 .maxScalar(0, s64)
872 .clampNumElements(0, v8s8, v16s8)
873 .clampNumElements(0, v4s16, v8s16)
874 .clampNumElements(0, v2s32, v4s32)
875 .clampMaxNumElements(0, s64, 2)
876 .lower();
877
878 // FP conversions
880 .legalFor(
881 {{f16, f32}, {f16, f64}, {f32, f64}, {v4f16, v4f32}, {v2f32, v2f64}})
882 .legalFor(ST.hasBF16(), {{bf16, f32}, {v4bf16, v4f32}})
883 .libcallFor({{f16, f128}, {f32, f128}, {f64, f128}})
885 .customIf([](const LegalityQuery &Q) {
886 LLT DstTy = Q.Types[0];
887 LLT SrcTy = Q.Types[1];
888 return SrcTy.getScalarSizeInBits() == 64 &&
889 DstTy.getScalarSizeInBits() == 16;
890 })
891 .lowerFor({{bf16, f32}, {v4bf16, v4f32}})
892 // Clamp based on input
893 .clampNumElements(1, v4s32, v4s32)
894 .clampNumElements(1, v2s64, v2s64)
895 .scalarize(0);
896
897 getActionDefinitionsBuilder(G_FPEXT)
898 .legalFor({{f32, f16},
899 {f64, f16},
900 {f32, bf16},
901 {f64, f32},
902 {v4f32, v4f16},
903 {v4f32, v4bf16},
904 {v2f64, v2f32}})
905 .libcallFor({{f128, f64}, {f128, f32}, {f128, f16}})
908 [](const LegalityQuery &Q) {
909 LLT DstTy = Q.Types[0];
910 LLT SrcTy = Q.Types[1];
911 return SrcTy.isVector() && DstTy.isVector() &&
912 SrcTy.getScalarSizeInBits() == 16 &&
913 DstTy.getScalarSizeInBits() == 64;
914 },
915 changeElementTo(1, f32))
916 .clampNumElements(0, v4s32, v4s32)
917 .clampNumElements(0, v2s64, v2s64)
918 .scalarize(0);
919
920 // Conversions
921 getActionDefinitionsBuilder({G_FPTOSI, G_FPTOUI})
922 .legalFor({{i32, f32},
923 {i64, f32},
924 {i32, f64},
925 {i64, f64},
926 {v2i32, v2f32},
927 {v4i32, v4f32},
928 {v2i64, v2f64}})
929 .legalFor(HasFP16,
930 {{i32, f16}, {i64, f16}, {v4i16, v4f16}, {v8i16, v8f16}})
931 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
933 // The range of a fp16 value fits into an i17, so we can lower the width
934 // to i64.
936 [=](const LegalityQuery &Query) {
937 return Query.Types[1] == f16 && Query.Types[0].getSizeInBits() > 64;
938 },
939 changeTo(0, i64))
942 .minScalar(0, s32)
944 [HasFP16](const LegalityQuery &Query) {
945 return (!HasFP16 && Query.Types[1].getScalarType().isFloat16()) ||
946 Query.Types[1].getScalarType().isBFloat16();
947 },
948 changeElementTo(1, f32))
949 .widenScalarIf(
950 [=](const LegalityQuery &Query) {
951 return Query.Types[0].getScalarSizeInBits() <= 64 &&
952 Query.Types[0].getScalarSizeInBits() >
953 Query.Types[1].getScalarSizeInBits();
954 },
956 .widenScalarIf(
957 [=](const LegalityQuery &Query) {
958 return Query.Types[1].getScalarSizeInBits() <= 64 &&
959 Query.Types[0].getScalarSizeInBits() <
960 Query.Types[1].getScalarSizeInBits();
961 },
963 .clampNumElements(0, v4s16, v8s16)
964 .clampNumElements(0, v2s32, v4s32)
965 .clampMaxNumElements(0, s64, 2)
966 .libcallFor(
967 {{i32, f128}, {i64, f128}, {i128, f128}, {i128, f32}, {i128, f64}});
968
969 getActionDefinitionsBuilder({G_FPTOSI_SAT, G_FPTOUI_SAT})
970 .legalFor({{i32, f32},
971 {i64, f32},
972 {i32, f64},
973 {i64, f64},
974 {v2i32, v2f32},
975 {v4i32, v4f32},
976 {v2i64, v2f64}})
977 .legalFor(
978 HasFP16,
979 {{i16, f16}, {i32, f16}, {i64, f16}, {v4i16, v4f16}, {v8i16, v8f16}})
980 // Handle types larger than i64 by scalarizing/lowering.
981 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
983 // The range of a fp16 value fits into an i17, so we can lower the width
984 // to i64.
986 [=](const LegalityQuery &Query) {
987 return Query.Types[1] == f16 && Query.Types[0].getSizeInBits() > 64;
988 },
989 changeTo(0, i64))
990 .lowerIf(::any(scalarWiderThan(0, 64), scalarWiderThan(1, 64)), 0)
992 .widenScalarToNextPow2(0, /*MinSize=*/32)
993 .minScalar(0, s32)
995 [HasFP16](const LegalityQuery &Query) {
996 return (!HasFP16 && Query.Types[1].getScalarType().isFloat16()) ||
997 Query.Types[1].getScalarType().isBFloat16();
998 },
999 changeElementTo(1, f32))
1000 .widenScalarIf(
1001 [=](const LegalityQuery &Query) {
1002 unsigned ITySize = Query.Types[0].getScalarSizeInBits();
1003 return (ITySize == 16 || ITySize == 32 || ITySize == 64) &&
1004 ITySize > Query.Types[1].getScalarSizeInBits();
1005 },
1007 .widenScalarIf(
1008 [=](const LegalityQuery &Query) {
1009 unsigned FTySize = Query.Types[1].getScalarSizeInBits();
1010 return (FTySize == 16 || FTySize == 32 || FTySize == 64) &&
1011 Query.Types[0].getScalarSizeInBits() < FTySize;
1012 },
1015 .clampNumElements(0, v4s16, v8s16)
1016 .clampNumElements(0, v2s32, v4s32)
1017 .clampMaxNumElements(0, s64, 2);
1018
1019 getActionDefinitionsBuilder({G_SITOFP, G_UITOFP})
1020 .legalFor({{f32, i32},
1021 {f64, i32},
1022 {f32, i64},
1023 {f64, i64},
1024 {v2f32, v2i32},
1025 {v4f32, v4i32},
1026 {v2f64, v2i64}})
1027 .legalFor(HasFP16,
1028 {{f16, i32}, {f16, i64}, {v4f16, v4i16}, {v8f16, v8i16}})
1029 .unsupportedIf([&](const LegalityQuery &Query) {
1030 return Query.Types[0].getScalarType().isBFloat16();
1031 })
1032 .scalarizeIf(scalarOrEltWiderThan(1, 64), 1)
1036 .minScalar(1, f32)
1037 .lowerIf([](const LegalityQuery &Query) {
1038 return Query.Types[1].isVector() &&
1039 Query.Types[1].getScalarSizeInBits() == 64 &&
1040 Query.Types[0].getScalarSizeInBits() == 16;
1041 })
1042 .widenScalarOrEltToNextPow2OrMinSize(0, /*MinSize=*/HasFP16 ? 16 : 32)
1043 .scalarizeIf(
1044 // v2i64->v2f32 needs to scalarize to avoid double-rounding issues.
1045 [](const LegalityQuery &Query) {
1046 return Query.Types[0].getScalarSizeInBits() == 32 &&
1047 Query.Types[1].getScalarSizeInBits() == 64;
1048 },
1049 0)
1050 .widenScalarIf(
1051 [](const LegalityQuery &Query) {
1052 return Query.Types[1].getScalarSizeInBits() <= 64 &&
1053 Query.Types[0].getScalarSizeInBits() <
1054 Query.Types[1].getScalarSizeInBits();
1055 },
1057 .widenScalarIf(
1058 [](const LegalityQuery &Query) {
1059 return Query.Types[0].getScalarSizeInBits() <= 64 &&
1060 Query.Types[0].getScalarSizeInBits() >
1061 Query.Types[1].getScalarSizeInBits();
1062 },
1064 .clampNumElements(0, v4s16, v8s16)
1065 .clampNumElements(0, v2s32, v4s32)
1066 .clampMaxNumElements(0, s64, 2)
1067 .libcallFor({{f16, i128},
1068 {f32, i128},
1069 {f64, i128},
1070 {f128, i128},
1071 {f128, i32},
1072 {f128, i64}});
1073
1074 // Control-flow
1075 getActionDefinitionsBuilder(G_BR).alwaysLegal();
1076 getActionDefinitionsBuilder(G_BRCOND)
1077 .legalFor({s32})
1078 .clampScalar(0, s32, s32);
1079 getActionDefinitionsBuilder(G_BRINDIRECT).legalFor({p0});
1080
1081 getActionDefinitionsBuilder(G_SELECT)
1082 .legalFor({{s32, s32}, {s64, s32}, {p0, s32}})
1083 .widenScalarToNextPow2(0)
1084 .clampScalar(0, s32, s64)
1085 .clampScalar(1, s32, s32)
1088 .lowerIf(isVector(0));
1089
1090 // Pointer-handling
1091 getActionDefinitionsBuilder(G_FRAME_INDEX).legalFor({p0});
1092
1093 if (TM.getCodeModel() == CodeModel::Small)
1094 getActionDefinitionsBuilder(G_GLOBAL_VALUE).custom();
1095 else
1096 getActionDefinitionsBuilder(G_GLOBAL_VALUE).legalFor({p0});
1097
1098 getActionDefinitionsBuilder(G_PTRAUTH_GLOBAL_VALUE)
1099 .legalIf(all(typeIs(0, p0), typeIs(1, p0)));
1100
1101 getActionDefinitionsBuilder(G_PTRTOINT)
1102 .legalFor({{i64, p0}, {v2i64, v2p0}})
1103 .widenScalarToNextPow2(0, 64)
1104 .clampScalar(0, s64, s64)
1105 .clampMaxNumElements(0, s64, 2);
1106
1107 getActionDefinitionsBuilder(G_INTTOPTR)
1108 .unsupportedIf([&](const LegalityQuery &Query) {
1109 return Query.Types[0].getSizeInBits() != Query.Types[1].getSizeInBits();
1110 })
1111 .legalFor({{p0, i64}, {v2p0, v2i64}})
1112 .clampMaxNumElements(1, s64, 2);
1113
1114 // Casts for 32 and 64-bit width type are just copies.
1115 // Same for 128-bit width type, except they are on the FPR bank.
1116 getActionDefinitionsBuilder(G_BITCAST)
1118 // Keeping 32-bit instructions legal to prevent regression in some tests
1119 .legalForCartesianProduct({s32, v2s16, v4s8})
1120 .legalForCartesianProduct({s64, v8s8, v4s16, v2s32})
1121 .legalForCartesianProduct({s128, v16s8, v8s16, v4s32, v2s64, v2p0})
1122 .customIf([=](const LegalityQuery &Query) {
1123 // Handle casts from i1 vectors to scalars.
1124 LLT DstTy = Query.Types[0];
1125 LLT SrcTy = Query.Types[1];
1126 return DstTy.isScalar() && SrcTy.isVector() &&
1127 SrcTy.getScalarSizeInBits() == 1;
1128 })
1129 .lowerIf([=](const LegalityQuery &Query) {
1130 return Query.Types[0].isVector() != Query.Types[1].isVector();
1131 })
1132 // moreElementsToNextPow2 cannot pad the source to match, so lower
1133 .lowerIf([=](const LegalityQuery &Query) {
1134 LLT DstTy = Query.Types[0];
1135 LLT SrcTy = Query.Types[1];
1136 if (!DstTy.isFixedVector() || !SrcTy.isFixedVector())
1137 return false;
1138 unsigned MoreElts = 1u << Log2_32_Ceil(DstTy.getNumElements());
1139 return SrcTy.getNumElements() * MoreElts % DstTy.getNumElements() != 0;
1140 })
1142 .clampNumElements(0, v8s8, v16s8)
1143 .clampNumElements(0, v4s16, v8s16)
1144 .clampNumElements(0, v2s32, v4s32)
1145 .clampMaxNumElements(0, s64, 2)
1146 .lower();
1147
1148 getActionDefinitionsBuilder(G_VASTART).legalFor({p0});
1149
1150 // va_list must be a pointer, but most sized types are pretty easy to handle
1151 // as the destination.
1152 getActionDefinitionsBuilder(G_VAARG)
1153 .customForCartesianProduct({s8, s16, s32, s64, p0}, {p0})
1154 .clampScalar(0, s8, s64)
1155 .widenScalarToNextPow2(0, /*Min*/ 8);
1156
1157 getActionDefinitionsBuilder(G_ATOMIC_CMPXCHG_WITH_SUCCESS)
1158 .lowerIf(
1159 all(typeInSet(0, {s8, s16, s32, s64, s128}), typeIs(2, p0)));
1160
1161 bool UseOutlineAtomics = ST.outlineAtomics() && !ST.hasLSE();
1162
1163 getActionDefinitionsBuilder(G_ATOMIC_CMPXCHG)
1164 .legalFor(!UseOutlineAtomics, {{s32, p0}, {s64, p0}})
1165 .customFor(!UseOutlineAtomics, {{s128, p0}})
1166 .libcallFor(UseOutlineAtomics,
1167 {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}, {s128, p0}})
1168 .clampScalar(0, s32, s64);
1169
1170 getActionDefinitionsBuilder({G_ATOMICRMW_XCHG, G_ATOMICRMW_ADD,
1171 G_ATOMICRMW_SUB, G_ATOMICRMW_AND, G_ATOMICRMW_OR,
1172 G_ATOMICRMW_XOR})
1173 .legalFor(!UseOutlineAtomics, {{s32, p0}, {s64, p0}})
1174 .libcallFor(UseOutlineAtomics,
1175 {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}})
1176 .clampScalar(0, s32, s64);
1177
1178 // Do not outline these atomics operations, as per comment in
1179 // AArch64ISelLowering.cpp's shouldExpandAtomicRMWInIR().
1180 getActionDefinitionsBuilder(
1181 {G_ATOMICRMW_MIN, G_ATOMICRMW_MAX, G_ATOMICRMW_UMIN, G_ATOMICRMW_UMAX})
1182 .legalIf(all(typeInSet(0, {s32, s64}), typeIs(1, p0)))
1183 .clampScalar(0, s32, s64);
1184
1185 getActionDefinitionsBuilder(G_BLOCK_ADDR).legalFor({p0});
1186
1187 // Merge/Unmerge
1188 for (unsigned Op : {G_MERGE_VALUES, G_UNMERGE_VALUES}) {
1189 unsigned BigTyIdx = Op == G_MERGE_VALUES ? 0 : 1;
1190 unsigned LitTyIdx = Op == G_MERGE_VALUES ? 1 : 0;
1191 getActionDefinitionsBuilder(Op)
1192 .widenScalarToNextPow2(LitTyIdx, 8)
1193 // Above s64 lowered shifts narrow back to a merge and never terminate
1194 .lowerIf([=](const LegalityQuery &Q) {
1195 const LLT BigTy = Q.Types[BigTyIdx];
1196 return BigTy.isScalar() && !isPowerOf2_32(BigTy.getSizeInBits()) &&
1197 BigTy.getSizeInBits() < 64;
1198 })
1199 .widenScalarToNextPow2(BigTyIdx, 32)
1200 .clampScalar(LitTyIdx, s8, s64)
1201 .clampScalar(BigTyIdx, s32, s128)
1202 .legalIf([=](const LegalityQuery &Q) {
1203 switch (Q.Types[BigTyIdx].getSizeInBits()) {
1204 case 32:
1205 case 64:
1206 case 128:
1207 break;
1208 default:
1209 return false;
1210 }
1211 switch (Q.Types[LitTyIdx].getSizeInBits()) {
1212 case 8:
1213 case 16:
1214 case 32:
1215 case 64:
1216 return true;
1217 default:
1218 return false;
1219 }
1220 });
1221 }
1222
1223 // TODO : nxv4s16, nxv2s16, nxv2s32
1224 getActionDefinitionsBuilder(G_EXTRACT_VECTOR_ELT)
1225 .legalFor(HasSVE, {{s16, nxv16s8, s64},
1226 {s16, nxv8s16, s64},
1227 {s32, nxv4s32, s64},
1228 {s64, nxv2s64, s64}})
1229 .unsupportedIf([=](const LegalityQuery &Query) {
1230 const LLT &EltTy = Query.Types[1].getElementType();
1231 if (Query.Types[1].isScalableVector())
1232 return false;
1233 return Query.Types[0] != EltTy;
1234 })
1235 .minScalar(2, s64)
1236 .customIf([=](const LegalityQuery &Query) {
1237 const LLT &VecTy = Query.Types[1];
1238 return VecTy == v8s8 || VecTy == v16s8 || VecTy == v2s16 ||
1239 VecTy == v4s16 || VecTy == v8s16 || VecTy == v2s32 ||
1240 VecTy == v4s32 || VecTy == v2s64 || VecTy == v2p0;
1241 })
1242 .minScalarOrEltIf(
1243 [=](const LegalityQuery &Query) {
1244 // We want to promote to <M x s1> to <M x s64> if that wouldn't
1245 // cause the total vec size to be > 128b.
1246 return Query.Types[1].isFixedVector() &&
1247 Query.Types[1].getNumElements() <= 2;
1248 },
1249 0, s64)
1250 .minScalarOrEltIf(
1251 [=](const LegalityQuery &Query) {
1252 return Query.Types[1].isFixedVector() &&
1253 Query.Types[1].getNumElements() <= 4;
1254 },
1255 0, s32)
1256 .minScalarOrEltIf(
1257 [=](const LegalityQuery &Query) {
1258 return Query.Types[1].isFixedVector() &&
1259 Query.Types[1].getNumElements() <= 8;
1260 },
1261 0, s16)
1262 .minScalarOrEltIf(
1263 [=](const LegalityQuery &Query) {
1264 return Query.Types[1].isFixedVector() &&
1265 Query.Types[1].getNumElements() <= 16;
1266 },
1267 0, s8)
1268 .minScalarOrElt(0, s8) // Worst case, we need at least s8.
1269 .moreElementsToNextPow2(1)
1270 .clampMaxNumElements(1, s64, 2)
1271 .clampMaxNumElements(1, s32, 4)
1272 .clampMaxNumElements(1, s16, 8)
1273 .clampMaxNumElements(1, s8, 16)
1274 .clampMaxNumElements(1, p0, 2)
1275 .scalarizeIf(scalarOrEltWiderThan(1, 64), 1);
1276
1277 getActionDefinitionsBuilder(G_INSERT_VECTOR_ELT)
1278 .legalIf(
1279 typeInSet(0, {v8s8, v16s8, v4s16, v8s16, v2s32, v4s32, v2s64, v2p0}))
1280 .legalFor(HasSVE, {{nxv16s8, s32, s64},
1281 {nxv8s16, s32, s64},
1282 {nxv4s32, s32, s64},
1283 {nxv2s64, s64, s64}})
1285 .widenVectorEltsToVectorMinSize(0, 64)
1286 .clampNumElements(0, v8s8, v16s8)
1287 .clampNumElements(0, v4s16, v8s16)
1288 .clampNumElements(0, v2s32, v4s32)
1289 .clampMaxNumElements(0, s64, 2)
1290 .clampMaxNumElements(0, p0, 2)
1291 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0);
1292
1293 getActionDefinitionsBuilder(G_BUILD_VECTOR)
1294 .legalFor({{v8s8, s8},
1295 {v16s8, s8},
1296 {v4s16, s16},
1297 {v8s16, s16},
1298 {v2s32, s32},
1299 {v4s32, s32},
1300 {v2s64, s64},
1301 {v2p0, p0}})
1302 .clampNumElements(0, v4s32, v4s32)
1303 .clampNumElements(0, v2s64, v2s64)
1304 .minScalarOrElt(0, s8)
1305 .widenVectorEltsToVectorMinSize(0, 64)
1306 .widenScalarOrEltToNextPow2(0)
1307 .minScalarSameAs(1, 0);
1308
1309 getActionDefinitionsBuilder(G_BUILD_VECTOR_TRUNC).lower();
1310
1311 getActionDefinitionsBuilder(G_SHUFFLE_VECTOR)
1312 .legalIf([=](const LegalityQuery &Query) {
1313 const LLT &DstTy = Query.Types[0];
1314 const LLT &SrcTy = Query.Types[1];
1315 // For now just support the TBL2 variant which needs the source vectors
1316 // to be the same size as the dest.
1317 if (DstTy != SrcTy)
1318 return false;
1319 return llvm::is_contained(
1320 {v8s8, v16s8, v4s16, v8s16, v2s32, v4s32, v2s64}, DstTy);
1321 })
1322 .moreElementsIf(
1323 [](const LegalityQuery &Query) {
1324 return Query.Types[0].getNumElements() >
1325 Query.Types[1].getNumElements();
1326 },
1327 changeTo(1, 0))
1329 .moreElementsIf(
1330 [](const LegalityQuery &Query) {
1331 return Query.Types[0].getNumElements() <
1332 Query.Types[1].getNumElements();
1333 },
1334 changeTo(0, 1))
1335 .widenScalarOrEltToNextPow2OrMinSize(0, 8)
1336 .clampNumElements(0, v8s8, v16s8)
1337 .clampNumElements(0, v4s16, v8s16)
1338 .clampNumElements(0, v4s32, v4s32)
1339 .clampNumElements(0, v2s64, v2s64)
1340 .scalarizeIf(scalarOrEltWiderThan(0, 64), 0)
1341 .bitcastIf(isPointerVector(0), [=](const LegalityQuery &Query) {
1342 // Bitcast pointers vector to i64.
1343 const LLT DstTy = Query.Types[0];
1344 return std::pair(
1345 0, LLT::vector(DstTy.getElementCount(), LLT::integer(64)));
1346 });
1347
1348 getActionDefinitionsBuilder(G_CONCAT_VECTORS)
1349 .legalFor({{v16s8, v8s8}, {v8s16, v4s16}, {v4s32, v2s32}})
1350 .customIf([=](const LegalityQuery &Query) {
1351 return Query.Types[0].isFixedVector() &&
1352 Query.Types[0].getScalarSizeInBits() < 8;
1353 })
1354 .bitcastIf(
1355 [=](const LegalityQuery &Query) {
1356 return Query.Types[0].isFixedVector() &&
1357 Query.Types[1].isFixedVector() &&
1358 Query.Types[0].getScalarSizeInBits() >= 8 &&
1359 isPowerOf2_64(Query.Types[0].getScalarSizeInBits()) &&
1360 Query.Types[0].getSizeInBits() <= 128 &&
1361 Query.Types[1].getSizeInBits() <= 64;
1362 },
1363 [=](const LegalityQuery &Query) {
1364 const LLT DstTy = Query.Types[0];
1365 const LLT SrcTy = Query.Types[1];
1366 return std::pair(
1367 0, DstTy.changeElementSize(SrcTy.getSizeInBits())
1370 SrcTy.getNumElements())));
1371 });
1372
1373 getActionDefinitionsBuilder(G_EXTRACT_SUBVECTOR)
1374 .legalFor({{v8s8, v16s8}, {v4s16, v8s16}, {v2s32, v4s32}})
1376 .clampMaxNumElements(0, s8, 16)
1377 .clampMaxNumElements(0, s16, 8)
1378 .clampMaxNumElements(0, s32, 4)
1379 .clampNumElements(1, v8s8, v16s8)
1380 .clampNumElements(1, v4s16, v8s16)
1381 .clampNumElements(1, v2s32, v4s32)
1382 .lower()
1383 .immIdx(0); // Inform verifier imm idx 0 is handled.
1384
1385 // TODO: {nxv16s8, s8}, {nxv8s16, s16}
1386 getActionDefinitionsBuilder(G_SPLAT_VECTOR)
1387 .legalFor(HasSVE, {{nxv4s32, s32}, {nxv2s64, s64}});
1388
1389 getActionDefinitionsBuilder(G_JUMP_TABLE).legalFor({p0});
1390
1391 getActionDefinitionsBuilder(G_BRJT).legalFor({{p0, s64}});
1392
1393 getActionDefinitionsBuilder({G_TRAP, G_DEBUGTRAP, G_UBSANTRAP}).alwaysLegal();
1394
1395 getActionDefinitionsBuilder(G_DYN_STACKALLOC).custom();
1396
1397 getActionDefinitionsBuilder({G_STACKSAVE, G_STACKRESTORE}).lower();
1398
1399 if (ST.hasMOPS()) {
1400 // G_BZERO is not supported. Currently it is only emitted by
1401 // PreLegalizerCombiner for G_MEMSET with zero constant.
1402 getActionDefinitionsBuilder(G_BZERO).unsupported();
1403
1404 getActionDefinitionsBuilder(G_MEMSET)
1405 .legalForCartesianProduct({p0}, {s64}, {s64})
1406 .customForCartesianProduct({p0}, {s8}, {s64})
1407 .immIdx(0); // Inform verifier imm idx 0 is handled.
1408
1409 getActionDefinitionsBuilder({G_MEMCPY, G_MEMMOVE})
1410 .legalForCartesianProduct({p0}, {p0}, {s64})
1411 .immIdx(0); // Inform verifier imm idx 0 is handled.
1412
1413 // G_MEMCPY_INLINE does not have a tailcall immediate
1414 getActionDefinitionsBuilder(G_MEMCPY_INLINE)
1415 .legalForCartesianProduct({p0}, {p0}, {s64});
1416
1417 getActionDefinitionsBuilder(G_MEMSET_INLINE)
1418 .legalForCartesianProduct({p0}, {s64}, {s64})
1419 .customForCartesianProduct({p0}, {s8}, {s64});
1420 } else {
1421 getActionDefinitionsBuilder({G_BZERO, G_MEMCPY, G_MEMMOVE, G_MEMSET})
1422 .libcall();
1423 }
1424
1425 // For fadd reductions we have pairwise operations available. We treat the
1426 // usual legal types as legal and handle the lowering to pairwise instructions
1427 // later.
1428 getActionDefinitionsBuilder(G_VECREDUCE_FADD)
1429 .legalFor({{f32, v2f32}, {f32, v4f32}, {f64, v2f64}})
1430 .legalFor(HasFP16, {{f16, v4f16}, {f16, v8f16}})
1431 .widenScalarIf(
1432 [HasFP16](const LegalityQuery &Query) {
1433 return (!HasFP16 && Query.Types[0].getScalarType().isFloat16()) ||
1434 Query.Types[0].getScalarType().isBFloat16();
1435 },
1436 changeElementTo(0, f32))
1437 .clampMaxNumElements(1, s64, 2)
1438 .clampMaxNumElements(1, s32, 4)
1439 .clampMaxNumElements(1, s16, 8)
1440 .moreElementsToNextPow2(1)
1441 .scalarize(1)
1442 .lower();
1443
1444 // For fmul reductions we need to split up into individual operations. We
1445 // clamp to 128 bit vectors then to 64bit vectors to produce a cascade of
1446 // smaller types, followed by scalarizing what remains.
1447 getActionDefinitionsBuilder(G_VECREDUCE_FMUL)
1448 .widenScalarIf(
1449 [HasFP16](const LegalityQuery &Query) {
1450 return (!HasFP16 && Query.Types[0].getScalarType().isFloat16()) ||
1451 Query.Types[0].getScalarType().isBFloat16();
1452 },
1453 changeElementTo(0, f32))
1454 .clampMaxNumElements(1, s64, 2)
1455 .clampMaxNumElements(1, s32, 4)
1456 .clampMaxNumElements(1, s16, 8)
1457 .clampMaxNumElements(1, s32, 2)
1458 .clampMaxNumElements(1, s16, 4)
1459 .scalarize(1)
1460 .lower();
1461
1462 getActionDefinitionsBuilder({G_VECREDUCE_SEQ_FADD, G_VECREDUCE_SEQ_FMUL})
1463 .scalarize(2)
1464 .lower();
1465
1466 getActionDefinitionsBuilder(G_VECREDUCE_ADD)
1467 .legalFor({{i8, v8i8},
1468 {i8, v16i8},
1469 {i16, v4i16},
1470 {i16, v8i16},
1471 {i32, v2i32},
1472 {i32, v4i32},
1473 {i64, v2i64}})
1475 .clampMaxNumElements(1, s64, 2)
1476 .clampMaxNumElements(1, s32, 4)
1477 .clampMaxNumElements(1, s16, 8)
1478 .clampMaxNumElements(1, s8, 16)
1479 .widenVectorEltsToVectorMinSize(1, 64)
1480 .scalarize(1);
1481
1482 getActionDefinitionsBuilder({G_VECREDUCE_FMIN, G_VECREDUCE_FMAX,
1483 G_VECREDUCE_FMINIMUM, G_VECREDUCE_FMAXIMUM})
1484 .legalFor({{f32, v2f32}, {f32, v4f32}, {f64, v2f64}})
1485 .legalFor(HasFP16, {{f16, v4f16}, {f16, v8f16}})
1486 .widenScalarIf(
1487 [HasFP16](const LegalityQuery &Query) {
1488 return (!HasFP16 && Query.Types[0].getScalarType().isFloat16()) ||
1489 Query.Types[0].getScalarType().isBFloat16();
1490 },
1491 changeElementTo(0, f32))
1492 .clampMaxNumElements(1, s64, 2)
1493 .clampMaxNumElements(1, s32, 4)
1494 .clampMaxNumElements(1, s16, 8)
1495 .scalarize(1)
1496 .lower();
1497
1498 getActionDefinitionsBuilder(G_VECREDUCE_MUL)
1499 .clampMaxNumElements(1, s32, 2)
1500 .clampMaxNumElements(1, s16, 4)
1501 .clampMaxNumElements(1, s8, 8)
1502 .scalarize(1)
1503 .lower();
1504
1505 getActionDefinitionsBuilder(
1506 {G_VECREDUCE_SMIN, G_VECREDUCE_SMAX, G_VECREDUCE_UMIN, G_VECREDUCE_UMAX})
1507 .legalFor({{i8, v8i8},
1508 {i8, v16i8},
1509 {i16, v4i16},
1510 {i16, v8i16},
1511 {i32, v2i32},
1512 {i32, v4i32}})
1513 .moreElementsIf(
1514 [=](const LegalityQuery &Query) {
1515 return Query.Types[1].isVector() &&
1516 Query.Types[1].getElementType() != s8 &&
1517 Query.Types[1].getNumElements() & 1;
1518 },
1520 .clampMaxNumElements(1, s64, 2)
1521 .clampMaxNumElements(1, s32, 4)
1522 .clampMaxNumElements(1, s16, 8)
1523 .clampMaxNumElements(1, s8, 16)
1524 .scalarize(1)
1525 .lower();
1526
1527 getActionDefinitionsBuilder(
1528 {G_VECREDUCE_OR, G_VECREDUCE_AND, G_VECREDUCE_XOR})
1529 // Try to break down into smaller vectors as long as they're at least 64
1530 // bits. This lets us use vector operations for some parts of the
1531 // reduction.
1532 .fewerElementsIf(
1533 [=](const LegalityQuery &Q) {
1534 LLT SrcTy = Q.Types[1];
1535 if (SrcTy.isScalar())
1536 return false;
1537 if (!isPowerOf2_32(SrcTy.getNumElements()))
1538 return false;
1539 // We can usually perform 64b vector operations.
1540 return SrcTy.getSizeInBits() > 64;
1541 },
1542 [=](const LegalityQuery &Q) {
1543 LLT SrcTy = Q.Types[1];
1544 return std::make_pair(1, SrcTy.divide(2));
1545 })
1546 .scalarize(1)
1547 .lower();
1548
1549 // TODO: Update this to correct handling when adding AArch64/SVE support.
1550 getActionDefinitionsBuilder(G_VECTOR_COMPRESS).lower();
1551
1552 // Access to floating-point environment.
1553 getActionDefinitionsBuilder({G_GET_FPENV, G_SET_FPENV, G_RESET_FPENV,
1554 G_GET_FPMODE, G_SET_FPMODE, G_RESET_FPMODE})
1555 .libcall();
1556
1557 getActionDefinitionsBuilder({G_GET_ROUNDING, G_SET_ROUNDING})
1558 .customFor({s32});
1559
1560 getActionDefinitionsBuilder(G_IS_FPCLASS).lower();
1561
1562 getActionDefinitionsBuilder(G_PREFETCH).custom();
1563
1564 getActionDefinitionsBuilder({G_SCMP, G_UCMP}).lower();
1565
1566 getActionDefinitionsBuilder({G_INTRINSIC, G_INTRINSIC_W_SIDE_EFFECTS})
1567 .alwaysLegal();
1568 getActionDefinitionsBuilder(G_FENCE).alwaysLegal();
1569 getActionDefinitionsBuilder(G_INVOKE_REGION_START).alwaysLegal();
1570
1571 verify(*ST.getInstrInfo());
1572}
1573
1576 LostDebugLocObserver &LocObserver) const {
1577 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
1578 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
1579 GISelChangeObserver &Observer = Helper.Observer;
1580 switch (MI.getOpcode()) {
1581 default:
1582 // No idea what to do.
1583 return false;
1584 case TargetOpcode::G_VAARG:
1585 return legalizeVaArg(MI, MRI, MIRBuilder);
1586 case TargetOpcode::G_LOAD:
1587 case TargetOpcode::G_STORE:
1588 return legalizeLoadStore(MI, MRI, MIRBuilder, Observer);
1589 case TargetOpcode::G_SHL:
1590 case TargetOpcode::G_ASHR:
1591 case TargetOpcode::G_LSHR:
1592 return legalizeShlAshrLshr(MI, MRI, MIRBuilder, Observer);
1593 case TargetOpcode::G_GLOBAL_VALUE:
1594 return legalizeSmallCMGlobalValue(MI, MRI, MIRBuilder, Observer);
1595 case TargetOpcode::G_SBFX:
1596 case TargetOpcode::G_UBFX:
1597 return legalizeBitfieldExtract(MI, MRI, Helper);
1598 case TargetOpcode::G_FSHL:
1599 case TargetOpcode::G_FSHR:
1600 return legalizeFunnelShift(MI, MRI, MIRBuilder, Observer, Helper);
1601 case TargetOpcode::G_ROTR:
1602 return legalizeRotate(MI, MRI, Helper);
1603 case TargetOpcode::G_CTPOP:
1604 return legalizeCTPOP(MI, MRI, Helper);
1605 case TargetOpcode::G_ATOMIC_CMPXCHG:
1606 return legalizeAtomicCmpxchg128(MI, MRI, Helper);
1607 case TargetOpcode::G_CTTZ:
1608 return legalizeCTTZ(MI, Helper);
1609 case TargetOpcode::G_BZERO:
1610 case TargetOpcode::G_MEMCPY:
1611 case TargetOpcode::G_MEMMOVE:
1612 case TargetOpcode::G_MEMSET:
1613 case TargetOpcode::G_MEMSET_INLINE:
1614 return legalizeMemOps(MI, Helper);
1615 case TargetOpcode::G_EXTRACT_VECTOR_ELT:
1616 return legalizeExtractVectorElt(MI, MRI, Helper);
1617 case TargetOpcode::G_DYN_STACKALLOC:
1618 return legalizeDynStackAlloc(MI, Helper);
1619 case TargetOpcode::G_PREFETCH:
1620 return legalizePrefetch(MI, Helper);
1621 case TargetOpcode::G_ABS:
1622 return Helper.lowerAbsToCNeg(MI);
1623 case TargetOpcode::G_ICMP:
1624 return legalizeICMP(MI, MRI, MIRBuilder);
1625 case TargetOpcode::G_BITCAST:
1626 return legalizeBitcast(MI, Helper);
1627 case TargetOpcode::G_CONCAT_VECTORS:
1628 return legalizeConcatVectors(MI, MRI, MIRBuilder);
1629 case TargetOpcode::G_FPTRUNC:
1630 // In order to lower f16 to f64 properly, we need to use f32 as an
1631 // intermediary
1632 return legalizeFptrunc(MI, MIRBuilder, MRI);
1633 case TargetOpcode::G_GET_ROUNDING:
1634 return legalizeGetRounding(MI, MIRBuilder, MRI, Helper);
1635 case TargetOpcode::G_SET_ROUNDING:
1636 return legalizeSetRounding(MI, MIRBuilder, MRI, Helper);
1637 }
1638
1639 llvm_unreachable("expected switch to return");
1640}
1641
1642bool AArch64LegalizerInfo::legalizeBitcast(MachineInstr &MI,
1643 LegalizerHelper &Helper) const {
1644 assert(MI.getOpcode() == TargetOpcode::G_BITCAST && "Unexpected opcode");
1645 auto [DstReg, DstTy, SrcReg, SrcTy] = MI.getFirst2RegLLTs();
1646 // We're trying to handle casts from i1 vectors to scalars but reloading from
1647 // stack.
1648 if (!DstTy.isScalar() || !SrcTy.isVector() ||
1649 SrcTy.getElementType() != LLT::scalar(1))
1650 return false;
1651
1652 Helper.createStackStoreLoad(DstReg, SrcReg);
1653 MI.eraseFromParent();
1654 return true;
1655}
1656
1657bool AArch64LegalizerInfo::legalizeFunnelShift(MachineInstr &MI,
1659 MachineIRBuilder &MIRBuilder,
1660 GISelChangeObserver &Observer,
1661 LegalizerHelper &Helper) const {
1662 assert(MI.getOpcode() == TargetOpcode::G_FSHL ||
1663 MI.getOpcode() == TargetOpcode::G_FSHR);
1664
1665 // Keep as G_FSHR if shift amount is a G_CONSTANT, else use generic
1666 // lowering
1667 Register ShiftNo = MI.getOperand(3).getReg();
1668 LLT ShiftTy = MRI.getType(ShiftNo);
1669 auto VRegAndVal = getIConstantVRegValWithLookThrough(ShiftNo, MRI);
1670
1671 // Adjust shift amount according to Opcode (FSHL/FSHR)
1672 // Convert FSHL to FSHR
1673 LLT OperationTy = MRI.getType(MI.getOperand(0).getReg());
1674 APInt BitWidth(ShiftTy.getSizeInBits(), OperationTy.getSizeInBits(), false);
1675
1676 // Lower non-constant shifts and leave zero shifts to the optimizer.
1677 if (!VRegAndVal || VRegAndVal->Value.urem(BitWidth) == 0)
1678 return (Helper.lowerFunnelShiftAsShifts(MI) ==
1680
1681 APInt Amount = VRegAndVal->Value.urem(BitWidth);
1682
1683 Amount = MI.getOpcode() == TargetOpcode::G_FSHL ? BitWidth - Amount : Amount;
1684
1685 // If the instruction is G_FSHR, has a 64-bit G_CONSTANT for shift amount
1686 // in the range of 0 <-> BitWidth, it is legal
1687 if (ShiftTy.getSizeInBits() == 64 && MI.getOpcode() == TargetOpcode::G_FSHR &&
1688 VRegAndVal->Value.ult(BitWidth))
1689 return true;
1690
1691 // Cast the ShiftNumber to a 64-bit type
1692 auto Cast64 = MIRBuilder.buildConstant(LLT::integer(64), Amount.zext(64));
1693
1694 if (MI.getOpcode() == TargetOpcode::G_FSHR) {
1695 Observer.changingInstr(MI);
1696 MI.getOperand(3).setReg(Cast64.getReg(0));
1697 Observer.changedInstr(MI);
1698 }
1699 // If Opcode is FSHL, remove the FSHL instruction and create a FSHR
1700 // instruction
1701 else if (MI.getOpcode() == TargetOpcode::G_FSHL) {
1702 MIRBuilder.buildInstr(TargetOpcode::G_FSHR, {MI.getOperand(0).getReg()},
1703 {MI.getOperand(1).getReg(), MI.getOperand(2).getReg(),
1704 Cast64.getReg(0)});
1705 MI.eraseFromParent();
1706 }
1707 return true;
1708}
1709
1710bool AArch64LegalizerInfo::legalizeICMP(MachineInstr &MI,
1712 MachineIRBuilder &MIRBuilder) const {
1713 Register DstReg = MI.getOperand(0).getReg();
1714 Register SrcReg1 = MI.getOperand(2).getReg();
1715 Register SrcReg2 = MI.getOperand(3).getReg();
1716 LLT DstTy = MRI.getType(DstReg);
1717 LLT SrcTy = MRI.getType(SrcReg1);
1718
1719 // Check the vector types are legal
1720 if (DstTy.getScalarSizeInBits() != SrcTy.getScalarSizeInBits() ||
1721 DstTy.getNumElements() != SrcTy.getNumElements() ||
1722 (DstTy.getSizeInBits() != 64 && DstTy.getSizeInBits() != 128))
1723 return false;
1724
1725 // Lowers G_ICMP NE => G_ICMP EQ to allow better pattern matching for
1726 // following passes
1727 CmpInst::Predicate Pred = (CmpInst::Predicate)MI.getOperand(1).getPredicate();
1728 if (Pred != CmpInst::ICMP_NE)
1729 return true;
1730 Register CmpReg =
1731 MIRBuilder
1732 .buildICmp(CmpInst::ICMP_EQ, MRI.getType(DstReg), SrcReg1, SrcReg2)
1733 .getReg(0);
1734 MIRBuilder.buildNot(DstReg, CmpReg);
1735
1736 MI.eraseFromParent();
1737 return true;
1738}
1739
1740bool AArch64LegalizerInfo::legalizeRotate(MachineInstr &MI,
1742 LegalizerHelper &Helper) const {
1743 // To allow for imported patterns to match, we ensure that the rotate amount
1744 // is 64b with an extension.
1745 Register AmtReg = MI.getOperand(2).getReg();
1746 LLT AmtTy = MRI.getType(AmtReg);
1747 (void)AmtTy;
1748 assert(AmtTy.isScalar() && "Expected a scalar rotate");
1749 assert(AmtTy.getSizeInBits() < 64 && "Expected this rotate to be legal");
1750 auto NewAmt = Helper.MIRBuilder.buildZExt(LLT::integer(64), AmtReg);
1751 Helper.Observer.changingInstr(MI);
1752 MI.getOperand(2).setReg(NewAmt.getReg(0));
1753 Helper.Observer.changedInstr(MI);
1754 return true;
1755}
1756
1757bool AArch64LegalizerInfo::legalizeSmallCMGlobalValue(
1759 GISelChangeObserver &Observer) const {
1760 assert(MI.getOpcode() == TargetOpcode::G_GLOBAL_VALUE);
1761 // We do this custom legalization to convert G_GLOBAL_VALUE into target ADRP +
1762 // G_ADD_LOW instructions.
1763 // By splitting this here, we can optimize accesses in the small code model by
1764 // folding in the G_ADD_LOW into the load/store offset.
1765 auto &GlobalOp = MI.getOperand(1);
1766 // Don't modify an intrinsic call.
1767 if (GlobalOp.isSymbol())
1768 return true;
1769 const auto* GV = GlobalOp.getGlobal();
1770 if (GV->isThreadLocal())
1771 return true; // Don't want to modify TLS vars.
1772
1773 auto &TM = ST->getTargetLowering()->getTargetMachine();
1774 unsigned OpFlags = ST->ClassifyGlobalReference(GV, TM);
1775
1776 if (OpFlags & AArch64II::MO_GOT)
1777 return true;
1778
1779 auto Offset = GlobalOp.getOffset();
1780 Register DstReg = MI.getOperand(0).getReg();
1781 auto ADRP = MIRBuilder.buildInstr(AArch64::ADRP, {LLT::pointer(0, 64)}, {})
1782 .addGlobalAddress(GV, Offset, OpFlags | AArch64II::MO_PAGE);
1783 // Set the regclass on the dest reg too.
1784 MRI.setRegClass(ADRP.getReg(0), &AArch64::GPR64RegClass);
1785
1786 // MO_TAGGED on the page indicates a tagged address. Set the tag now. We do so
1787 // by creating a MOVK that sets bits 48-63 of the register to (global address
1788 // + 0x100000000 - PC) >> 48. The additional 0x100000000 offset here is to
1789 // prevent an incorrect tag being generated during relocation when the
1790 // global appears before the code section. Without the offset, a global at
1791 // `0x0f00'0000'0000'1000` (i.e. at `0x1000` with tag `0xf`) that's referenced
1792 // by code at `0x2000` would result in `0x0f00'0000'0000'1000 - 0x2000 =
1793 // 0x0eff'ffff'ffff'f000`, meaning the tag would be incorrectly set to `0xe`
1794 // instead of `0xf`.
1795 // This assumes that we're in the small code model so we can assume a binary
1796 // size of <= 4GB, which makes the untagged PC relative offset positive. The
1797 // binary must also be loaded into address range [0, 2^48). Both of these
1798 // properties need to be ensured at runtime when using tagged addresses.
1799 if (OpFlags & AArch64II::MO_TAGGED) {
1800 assert(!Offset &&
1801 "Should not have folded in an offset for a tagged global!");
1802 ADRP = MIRBuilder.buildInstr(AArch64::MOVKXi, {LLT::pointer(0, 64)}, {ADRP})
1803 .addGlobalAddress(GV, 0x100000000,
1805 .addImm(48);
1806 MRI.setRegClass(ADRP.getReg(0), &AArch64::GPR64RegClass);
1807 }
1808
1809 MIRBuilder.buildInstr(AArch64::G_ADD_LOW, {DstReg}, {ADRP})
1810 .addGlobalAddress(GV, Offset,
1812 MI.eraseFromParent();
1813 return true;
1814}
1815
1817 MachineInstr &MI) const {
1818 MachineIRBuilder &MIB = Helper.MIRBuilder;
1819 MachineRegisterInfo &MRI = *MIB.getMRI();
1820
1821 auto LowerUnaryOp = [&MI, &MIB](unsigned Opcode) {
1822 MIB.buildInstr(Opcode, {MI.getOperand(0)}, {MI.getOperand(2)});
1823 MI.eraseFromParent();
1824 return true;
1825 };
1826 auto LowerBinOp = [&MI, &MIB](unsigned Opcode) {
1827 MIB.buildInstr(Opcode, {MI.getOperand(0)},
1828 {MI.getOperand(2), MI.getOperand(3)});
1829 MI.eraseFromParent();
1830 return true;
1831 };
1832 auto LowerTriOp = [&MI, &MIB](unsigned Opcode) {
1833 MIB.buildInstr(Opcode, {MI.getOperand(0)},
1834 {MI.getOperand(2), MI.getOperand(3), MI.getOperand(4)});
1835 MI.eraseFromParent();
1836 return true;
1837 };
1838
1839 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
1840 switch (IntrinsicID) {
1841 case Intrinsic::vacopy: {
1842 unsigned PtrSize = ST->isTargetILP32() ? 4 : 8;
1843 unsigned VaListSize =
1844 (ST->isTargetDarwin() || ST->isTargetWindows())
1845 ? PtrSize
1846 : ST->isTargetILP32() ? 20 : 32;
1847
1848 MachineFunction &MF = *MI.getMF();
1850 LLT::integer(VaListSize * 8));
1851 MIB.buildLoad(Val, MI.getOperand(2),
1854 VaListSize, Align(PtrSize)));
1855 MIB.buildStore(Val, MI.getOperand(1),
1858 VaListSize, Align(PtrSize)));
1859 MI.eraseFromParent();
1860 return true;
1861 }
1862 case Intrinsic::get_dynamic_area_offset: {
1863 MIB.buildConstant(MI.getOperand(0).getReg(), 0);
1864 MI.eraseFromParent();
1865 return true;
1866 }
1867 case Intrinsic::aarch64_mops_memset_tag: {
1868 assert(MI.getOpcode() == TargetOpcode::G_INTRINSIC_W_SIDE_EFFECTS);
1869 // Anyext the value being set to 64 bit (only the bottom 8 bits are read by
1870 // the instruction).
1871 auto &Value = MI.getOperand(3);
1872 Register ExtValueReg = MIB.buildAnyExt(LLT::integer(64), Value).getReg(0);
1873 Value.setReg(ExtValueReg);
1874 return true;
1875 }
1876 case Intrinsic::aarch64_prefetch: {
1877 auto &AddrVal = MI.getOperand(1);
1878
1879 int64_t IsWrite = MI.getOperand(2).getImm();
1880 int64_t Target = MI.getOperand(3).getImm();
1881 int64_t IsStream = MI.getOperand(4).getImm();
1882 int64_t IsData = MI.getOperand(5).getImm();
1883
1884 unsigned PrfOp = (IsWrite << 4) | // Load/Store bit
1885 (!IsData << 3) | // IsDataCache bit
1886 (Target << 1) | // Cache level bits
1887 (unsigned)IsStream; // Stream bit
1888
1889 MIB.buildInstr(AArch64::G_AARCH64_PREFETCH).addImm(PrfOp).add(AddrVal);
1890 MI.eraseFromParent();
1891 return true;
1892 }
1893 case Intrinsic::aarch64_range_prefetch: {
1894 auto &AddrVal = MI.getOperand(1);
1895
1896 int64_t IsWrite = MI.getOperand(2).getImm();
1897 int64_t IsStream = MI.getOperand(3).getImm();
1898 unsigned PrfOp = (IsStream << 2) | IsWrite;
1899
1900 MIB.buildInstr(AArch64::G_AARCH64_RANGE_PREFETCH)
1901 .addImm(PrfOp)
1902 .add(AddrVal)
1903 .addUse(MI.getOperand(4).getReg()); // Metadata
1904 MI.eraseFromParent();
1905 return true;
1906 }
1907 case Intrinsic::aarch64_prefetch_ir: {
1908 auto &AddrVal = MI.getOperand(1);
1909 MIB.buildInstr(AArch64::G_AARCH64_PREFETCH).addImm(24).add(AddrVal);
1910 MI.eraseFromParent();
1911 return true;
1912 }
1913 case Intrinsic::aarch64_neon_uaddv:
1914 case Intrinsic::aarch64_neon_saddv:
1915 case Intrinsic::aarch64_neon_umaxv:
1916 case Intrinsic::aarch64_neon_smaxv:
1917 case Intrinsic::aarch64_neon_uminv:
1918 case Intrinsic::aarch64_neon_sminv: {
1919 bool IsSigned = IntrinsicID == Intrinsic::aarch64_neon_saddv ||
1920 IntrinsicID == Intrinsic::aarch64_neon_smaxv ||
1921 IntrinsicID == Intrinsic::aarch64_neon_sminv;
1922
1923 auto OldDst = MI.getOperand(0).getReg();
1924 auto OldDstTy = MRI.getType(OldDst);
1925 LLT NewDstTy = MRI.getType(MI.getOperand(2).getReg()).getElementType();
1926 if (OldDstTy == NewDstTy)
1927 return true;
1928
1929 auto NewDst = MRI.createGenericVirtualRegister(NewDstTy);
1930
1931 Helper.Observer.changingInstr(MI);
1932 MI.getOperand(0).setReg(NewDst);
1933 Helper.Observer.changedInstr(MI);
1934
1935 MIB.setInsertPt(MIB.getMBB(), ++MIB.getInsertPt());
1936 MIB.buildExtOrTrunc(IsSigned ? TargetOpcode::G_SEXT : TargetOpcode::G_ZEXT,
1937 OldDst, NewDst);
1938
1939 return true;
1940 }
1941 case Intrinsic::aarch64_neon_uaddlp:
1942 case Intrinsic::aarch64_neon_saddlp: {
1943 unsigned Opc = IntrinsicID == Intrinsic::aarch64_neon_uaddlp
1944 ? AArch64::G_UADDLP
1945 : AArch64::G_SADDLP;
1946 MIB.buildInstr(Opc, {MI.getOperand(0)}, {MI.getOperand(2)});
1947 MI.eraseFromParent();
1948
1949 return true;
1950 }
1951 case Intrinsic::aarch64_neon_uaddlv:
1952 case Intrinsic::aarch64_neon_saddlv: {
1953 unsigned Opc = IntrinsicID == Intrinsic::aarch64_neon_uaddlv
1954 ? AArch64::G_UADDLV
1955 : AArch64::G_SADDLV;
1956 Register DstReg = MI.getOperand(0).getReg();
1957 Register SrcReg = MI.getOperand(2).getReg();
1958 LLT DstTy = MRI.getType(DstReg);
1959
1960 LLT MidTy, ExtTy;
1961 if (DstTy.isScalar() && DstTy.getScalarSizeInBits() <= 32) {
1962 ExtTy = LLT::integer(32);
1963 MidTy = LLT::fixed_vector(4, ExtTy);
1964 } else {
1965 ExtTy = LLT::integer(64);
1966 MidTy = LLT::fixed_vector(2, ExtTy);
1967 }
1968
1969 Register MidReg =
1970 MIB.buildInstr(Opc, {MidTy}, {SrcReg})->getOperand(0).getReg();
1971 Register ZeroReg =
1972 MIB.buildConstant(LLT::integer(64), 0)->getOperand(0).getReg();
1973 Register ExtReg = MIB.buildInstr(AArch64::G_EXTRACT_VECTOR_ELT, {ExtTy},
1974 {MidReg, ZeroReg})
1975 .getReg(0);
1976
1977 if (DstTy.getScalarSizeInBits() < 32)
1978 MIB.buildTrunc(DstReg, ExtReg);
1979 else
1980 MIB.buildCopy(DstReg, ExtReg);
1981
1982 MI.eraseFromParent();
1983
1984 return true;
1985 }
1986 case Intrinsic::aarch64_neon_fmax:
1987 return LowerBinOp(TargetOpcode::G_FMAXIMUM);
1988 case Intrinsic::aarch64_neon_fmin:
1989 return LowerBinOp(TargetOpcode::G_FMINIMUM);
1990 case Intrinsic::aarch64_neon_fmaxnm:
1991 return LowerBinOp(TargetOpcode::G_FMAXNUM);
1992 case Intrinsic::aarch64_neon_fminnm:
1993 return LowerBinOp(TargetOpcode::G_FMINNUM);
1994 case Intrinsic::aarch64_neon_pmul:
1995 return LowerBinOp(TargetOpcode::G_CLMUL);
1996 case Intrinsic::aarch64_neon_pmull:
1997 case Intrinsic::aarch64_neon_pmull64:
1998 return LowerBinOp(AArch64::G_PMULL);
1999 case Intrinsic::aarch64_neon_smull:
2000 return LowerBinOp(AArch64::G_SMULL);
2001 case Intrinsic::aarch64_neon_umull:
2002 return LowerBinOp(AArch64::G_UMULL);
2003 case Intrinsic::aarch64_neon_sabd:
2004 return LowerBinOp(TargetOpcode::G_ABDS);
2005 case Intrinsic::aarch64_neon_uabd:
2006 return LowerBinOp(TargetOpcode::G_ABDU);
2007 case Intrinsic::aarch64_neon_uhadd:
2008 return LowerBinOp(TargetOpcode::G_UAVGFLOOR);
2009 case Intrinsic::aarch64_neon_urhadd:
2010 return LowerBinOp(TargetOpcode::G_UAVGCEIL);
2011 case Intrinsic::aarch64_neon_shadd:
2012 return LowerBinOp(TargetOpcode::G_SAVGFLOOR);
2013 case Intrinsic::aarch64_neon_srhadd:
2014 return LowerBinOp(TargetOpcode::G_SAVGCEIL);
2015 case Intrinsic::aarch64_neon_sqshrn: {
2016 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2017 return true;
2018 // Create right shift instruction. Store the output register in Shr.
2019 auto Shr = MIB.buildInstr(AArch64::G_VASHR,
2020 {MRI.getType(MI.getOperand(2).getReg())},
2021 {MI.getOperand(2), MI.getOperand(3).getImm()});
2022 // Build the narrow intrinsic, taking in Shr.
2023 MIB.buildInstr(TargetOpcode::G_TRUNC_SSAT_S, {MI.getOperand(0)}, {Shr});
2024 MI.eraseFromParent();
2025 return true;
2026 }
2027 case Intrinsic::aarch64_neon_sqshrun: {
2028 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2029 return true;
2030 // Create right shift instruction. Store the output register in Shr.
2031 auto Shr = MIB.buildInstr(AArch64::G_VASHR,
2032 {MRI.getType(MI.getOperand(2).getReg())},
2033 {MI.getOperand(2), MI.getOperand(3).getImm()});
2034 // Build the narrow intrinsic, taking in Shr.
2035 MIB.buildInstr(TargetOpcode::G_TRUNC_SSAT_U, {MI.getOperand(0)}, {Shr});
2036 MI.eraseFromParent();
2037 return true;
2038 }
2039 case Intrinsic::aarch64_neon_sqrshrn: {
2040 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2041 return true;
2042 // Create right shift instruction. Store the output register in Shr.
2043 auto Shr = MIB.buildInstr(AArch64::G_SRSHR_I,
2044 {MRI.getType(MI.getOperand(2).getReg())},
2045 {MI.getOperand(2), MI.getOperand(3).getImm()});
2046 // Build the narrow intrinsic, taking in Shr.
2047 MIB.buildInstr(TargetOpcode::G_TRUNC_SSAT_S, {MI.getOperand(0)}, {Shr});
2048 MI.eraseFromParent();
2049 return true;
2050 }
2051 case Intrinsic::aarch64_neon_sqrshrun: {
2052 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2053 return true;
2054 // Create right shift instruction. Store the output register in Shr.
2055 auto Shr = MIB.buildInstr(AArch64::G_SRSHR_I,
2056 {MRI.getType(MI.getOperand(2).getReg())},
2057 {MI.getOperand(2), MI.getOperand(3).getImm()});
2058 // Build the narrow intrinsic, taking in Shr.
2059 MIB.buildInstr(TargetOpcode::G_TRUNC_SSAT_U, {MI.getOperand(0)}, {Shr});
2060 MI.eraseFromParent();
2061 return true;
2062 }
2063 case Intrinsic::aarch64_neon_uqrshrn: {
2064 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2065 return true;
2066 // Create right shift instruction. Store the output register in Shr.
2067 auto Shr = MIB.buildInstr(AArch64::G_URSHR_I,
2068 {MRI.getType(MI.getOperand(2).getReg())},
2069 {MI.getOperand(2), MI.getOperand(3).getImm()});
2070 // Build the narrow intrinsic, taking in Shr.
2071 MIB.buildInstr(TargetOpcode::G_TRUNC_USAT_U, {MI.getOperand(0)}, {Shr});
2072 MI.eraseFromParent();
2073 return true;
2074 }
2075 case Intrinsic::aarch64_neon_uqshrn: {
2076 if (!MRI.getType(MI.getOperand(0).getReg()).isVector())
2077 return true;
2078 // Create right shift instruction. Store the output register in Shr.
2079 auto Shr = MIB.buildInstr(AArch64::G_VLSHR,
2080 {MRI.getType(MI.getOperand(2).getReg())},
2081 {MI.getOperand(2), MI.getOperand(3).getImm()});
2082 // Build the narrow intrinsic, taking in Shr.
2083 MIB.buildInstr(TargetOpcode::G_TRUNC_USAT_U, {MI.getOperand(0)}, {Shr});
2084 MI.eraseFromParent();
2085 return true;
2086 }
2087 case Intrinsic::aarch64_neon_sqshlu: {
2088 // Check if last operand is constant vector dup
2089 auto ShiftAmount =
2090 isConstantOrConstantSplatVector(MI.getOperand(3).getReg(), MRI);
2091 if (ShiftAmount) {
2092 // If so, create a new intrinsic with the correct shift amount
2093 MIB.buildInstr(AArch64::G_SQSHLU_I, {MI.getOperand(0)},
2094 {MI.getOperand(2)})
2095 .addImm(ShiftAmount->getSExtValue());
2096 MI.eraseFromParent();
2097 return true;
2098 }
2099 return false;
2100 }
2101 case Intrinsic::aarch64_neon_vsli: {
2102 MIB.buildInstr(
2103 AArch64::G_SLI, {MI.getOperand(0)},
2104 {MI.getOperand(2), MI.getOperand(3), MI.getOperand(4).getImm()});
2105 MI.eraseFromParent();
2106 break;
2107 }
2108 case Intrinsic::aarch64_neon_vsri: {
2109 MIB.buildInstr(
2110 AArch64::G_SRI, {MI.getOperand(0)},
2111 {MI.getOperand(2), MI.getOperand(3), MI.getOperand(4).getImm()});
2112 MI.eraseFromParent();
2113 break;
2114 }
2115 case Intrinsic::aarch64_neon_abs: {
2116 // Lower the intrinsic to G_ABS.
2117 MIB.buildInstr(TargetOpcode::G_ABS, {MI.getOperand(0)}, {MI.getOperand(2)});
2118 MI.eraseFromParent();
2119 return true;
2120 }
2121 case Intrinsic::aarch64_neon_addhn:
2122 return LowerBinOp(AArch64::G_ADDHN);
2123 case Intrinsic::aarch64_neon_sqadd: {
2124 if (MRI.getType(MI.getOperand(0).getReg()).isVector())
2125 return LowerBinOp(TargetOpcode::G_SADDSAT);
2126 break;
2127 }
2128 case Intrinsic::aarch64_neon_sqsub: {
2129 if (MRI.getType(MI.getOperand(0).getReg()).isVector())
2130 return LowerBinOp(TargetOpcode::G_SSUBSAT);
2131 break;
2132 }
2133 case Intrinsic::aarch64_neon_uqadd: {
2134 if (MRI.getType(MI.getOperand(0).getReg()).isVector())
2135 return LowerBinOp(TargetOpcode::G_UADDSAT);
2136 break;
2137 }
2138 case Intrinsic::aarch64_neon_uqsub: {
2139 if (MRI.getType(MI.getOperand(0).getReg()).isVector())
2140 return LowerBinOp(TargetOpcode::G_USUBSAT);
2141 break;
2142 }
2143 case Intrinsic::aarch64_neon_udot:
2144 return LowerTriOp(AArch64::G_UDOT);
2145 case Intrinsic::aarch64_neon_sdot:
2146 return LowerTriOp(AArch64::G_SDOT);
2147 case Intrinsic::aarch64_neon_usdot:
2148 return LowerTriOp(AArch64::G_USDOT);
2149 case Intrinsic::aarch64_neon_sqxtn:
2150 return LowerUnaryOp(TargetOpcode::G_TRUNC_SSAT_S);
2151 case Intrinsic::aarch64_neon_sqxtun:
2152 return LowerUnaryOp(TargetOpcode::G_TRUNC_SSAT_U);
2153 case Intrinsic::aarch64_neon_uqxtn:
2154 return LowerUnaryOp(TargetOpcode::G_TRUNC_USAT_U);
2155 case Intrinsic::aarch64_neon_fcvtzu:
2156 return LowerUnaryOp(TargetOpcode::G_FPTOUI_SAT);
2157 case Intrinsic::aarch64_neon_fcvtzs:
2158 return LowerUnaryOp(TargetOpcode::G_FPTOSI_SAT);
2159 case Intrinsic::aarch64_neon_cls:
2160 return LowerUnaryOp(TargetOpcode::G_CTLS);
2161
2162 case Intrinsic::vector_reverse:
2163 // TODO: Add support for vector_reverse
2164 return false;
2165 }
2166
2167 return true;
2168}
2169
2170bool AArch64LegalizerInfo::legalizeShlAshrLshr(
2172 GISelChangeObserver &Observer) const {
2173 assert(MI.getOpcode() == TargetOpcode::G_ASHR ||
2174 MI.getOpcode() == TargetOpcode::G_LSHR ||
2175 MI.getOpcode() == TargetOpcode::G_SHL);
2176 // If the shift amount is a G_CONSTANT, promote it to a 64 bit type so the
2177 // imported patterns can select it later. Either way, it will be legal.
2178 Register AmtReg = MI.getOperand(2).getReg();
2179 LLT AmtRegEltTy = MRI.getType(AmtReg).getScalarType();
2180 auto VRegAndVal = getIConstantVRegValWithLookThrough(AmtReg, MRI);
2181 if (!VRegAndVal)
2182 return true;
2183 // Check the shift amount is in range for an immediate form.
2184 int64_t Amount = VRegAndVal->Value.getSExtValue();
2185 if (Amount > 31)
2186 return true; // This will have to remain a register variant.
2187 auto ExtCst =
2188 MIRBuilder.buildConstant(AmtRegEltTy.changeElementSize(64), Amount);
2189 Observer.changingInstr(MI);
2190 MI.getOperand(2).setReg(ExtCst.getReg(0));
2191 Observer.changedInstr(MI);
2192 return true;
2193}
2194
2196 MachineRegisterInfo &MRI) {
2197 Base = Root;
2198 Offset = 0;
2199
2200 Register NewBase;
2201 int64_t NewOffset;
2202 if (mi_match(Root, MRI, m_GPtrAdd(m_Reg(NewBase), m_ICst(NewOffset))) &&
2203 isShiftedInt<7, 3>(NewOffset)) {
2204 Base = NewBase;
2205 Offset = NewOffset;
2206 }
2207}
2208
2209// FIXME: This should be removed and replaced with the generic bitcast legalize
2210// action.
2211bool AArch64LegalizerInfo::legalizeLoadStore(
2213 GISelChangeObserver &Observer) const {
2214 assert(MI.getOpcode() == TargetOpcode::G_STORE ||
2215 MI.getOpcode() == TargetOpcode::G_LOAD);
2216 // Here we just try to handle vector loads/stores where our value type might
2217 // have pointer elements, which the SelectionDAG importer can't handle. To
2218 // allow the existing patterns for s64 to fire for p0, we just try to bitcast
2219 // the value to use s64 types.
2220
2221 // Custom legalization requires the instruction, if not deleted, must be fully
2222 // legalized. In order to allow further legalization of the inst, we create
2223 // a new instruction and erase the existing one.
2224
2225 Register ValReg = MI.getOperand(0).getReg();
2226 const LLT ValTy = MRI.getType(ValReg);
2227
2228 if (ValTy == LLT::scalar(128)) {
2229
2230 AtomicOrdering Ordering = (*MI.memoperands_begin())->getSuccessOrdering();
2231 bool IsLoad = MI.getOpcode() == TargetOpcode::G_LOAD;
2232 bool IsLoadAcquire = IsLoad && Ordering == AtomicOrdering::Acquire;
2233 bool IsStoreRelease = !IsLoad && Ordering == AtomicOrdering::Release;
2234 bool IsRcpC3 =
2235 ST->hasLSE2() && ST->hasRCPC3() && (IsLoadAcquire || IsStoreRelease);
2236
2237 LLT s64 = LLT::integer(64);
2238
2239 unsigned Opcode;
2240 if (IsRcpC3) {
2241 Opcode = IsLoad ? AArch64::LDIAPPX : AArch64::STILPX;
2242 } else {
2243 // For LSE2, loads/stores should have been converted to monotonic and had
2244 // a fence inserted after them.
2245 assert(Ordering == AtomicOrdering::Monotonic ||
2246 Ordering == AtomicOrdering::Unordered);
2247 assert(ST->hasLSE2() && "ldp/stp not single copy atomic without +lse2");
2248
2249 Opcode = IsLoad ? AArch64::LDPXi : AArch64::STPXi;
2250 }
2251
2252 MachineInstrBuilder NewI;
2253 if (IsLoad) {
2254 NewI = MIRBuilder.buildInstr(Opcode, {s64, s64}, {});
2255 MIRBuilder.buildMergeLikeInstr(
2256 ValReg, {NewI->getOperand(0), NewI->getOperand(1)});
2257 } else {
2258 auto Split = MIRBuilder.buildUnmerge(s64, MI.getOperand(0));
2259 NewI = MIRBuilder.buildInstr(
2260 Opcode, {}, {Split->getOperand(0), Split->getOperand(1)});
2261 }
2262
2263 if (IsRcpC3) {
2264 NewI.addUse(MI.getOperand(1).getReg());
2265 } else {
2266 Register Base;
2267 int Offset;
2268 matchLDPSTPAddrMode(MI.getOperand(1).getReg(), Base, Offset, MRI);
2269 NewI.addUse(Base);
2270 NewI.addImm(Offset / 8);
2271 }
2272
2273 NewI.cloneMemRefs(MI);
2274 constrainSelectedInstRegOperands(*NewI, *ST->getInstrInfo(),
2275 *MRI.getTargetRegisterInfo(),
2276 *ST->getRegBankInfo());
2277 MI.eraseFromParent();
2278 return true;
2279 }
2280
2281 if (!ValTy.isPointerVector() ||
2282 ValTy.getElementType().getAddressSpace() != 0) {
2283 LLVM_DEBUG(dbgs() << "Tried to do custom legalization on wrong load/store");
2284 return false;
2285 }
2286
2287 unsigned PtrSize = ValTy.getElementType().getSizeInBits();
2288 const LLT NewTy = LLT::vector(ValTy.getElementCount(), LLT::integer(PtrSize));
2289 auto &MMO = **MI.memoperands_begin();
2290 MMO.setType(NewTy);
2291
2292 if (MI.getOpcode() == TargetOpcode::G_STORE) {
2293 auto Bitcast = MIRBuilder.buildBitcast(NewTy, ValReg);
2294 MIRBuilder.buildStore(Bitcast.getReg(0), MI.getOperand(1), MMO);
2295 } else {
2296 auto NewLoad = MIRBuilder.buildLoad(NewTy, MI.getOperand(1), MMO);
2297 MIRBuilder.buildBitcast(ValReg, NewLoad);
2298 }
2299 MI.eraseFromParent();
2300 return true;
2301}
2302
2303bool AArch64LegalizerInfo::legalizeVaArg(MachineInstr &MI,
2305 MachineIRBuilder &MIRBuilder) const {
2306 MachineFunction &MF = MIRBuilder.getMF();
2307 Align Alignment(MI.getOperand(2).getImm());
2308 Register Dst = MI.getOperand(0).getReg();
2309 Register ListPtr = MI.getOperand(1).getReg();
2310
2311 LLT PtrTy = MRI.getType(ListPtr);
2312 LLT IntPtrTy = LLT::integer(PtrTy.getSizeInBits());
2313
2314 const unsigned PtrSize = PtrTy.getSizeInBits() / 8;
2315 const Align PtrAlign = Align(PtrSize);
2316 auto List = MIRBuilder.buildLoad(
2317 PtrTy, ListPtr,
2318 *MF.getMachineMemOperand(MachinePointerInfo(), MachineMemOperand::MOLoad,
2319 PtrTy, PtrAlign));
2320
2321 MachineInstrBuilder DstPtr;
2322 if (Alignment > PtrAlign) {
2323 // Realign the list to the actual required alignment.
2324 auto AlignMinus1 =
2325 MIRBuilder.buildConstant(IntPtrTy, Alignment.value() - 1);
2326 auto ListTmp = MIRBuilder.buildPtrAdd(PtrTy, List, AlignMinus1.getReg(0));
2327 DstPtr = MIRBuilder.buildMaskLowPtrBits(PtrTy, ListTmp, Log2(Alignment));
2328 } else
2329 DstPtr = List;
2330
2331 LLT ValTy = MRI.getType(Dst);
2332 uint64_t ValSize = ValTy.getSizeInBits() / 8;
2333 MIRBuilder.buildLoad(
2334 Dst, DstPtr,
2335 *MF.getMachineMemOperand(MachinePointerInfo(), MachineMemOperand::MOLoad,
2336 ValTy, std::max(Alignment, PtrAlign)));
2337
2338 auto Size = MIRBuilder.buildConstant(IntPtrTy, alignTo(ValSize, PtrAlign));
2339
2340 auto NewList = MIRBuilder.buildPtrAdd(PtrTy, DstPtr, Size.getReg(0));
2341
2342 MIRBuilder.buildStore(NewList, ListPtr,
2343 *MF.getMachineMemOperand(MachinePointerInfo(),
2345 PtrTy, PtrAlign));
2346
2347 MI.eraseFromParent();
2348 return true;
2349}
2350
2351bool AArch64LegalizerInfo::legalizeBitfieldExtract(
2352 MachineInstr &MI, MachineRegisterInfo &MRI, LegalizerHelper &Helper) const {
2353 // Only legal if we can select immediate forms.
2354 // TODO: Lower this otherwise.
2355 return getIConstantVRegValWithLookThrough(MI.getOperand(2).getReg(), MRI) &&
2356 getIConstantVRegValWithLookThrough(MI.getOperand(3).getReg(), MRI);
2357}
2358
2359bool AArch64LegalizerInfo::legalizeCTPOP(MachineInstr &MI,
2361 LegalizerHelper &Helper) const {
2362 // When there is no integer popcount instruction (FEAT_CSSC isn't available),
2363 // it can be more efficiently lowered to the following sequence that uses
2364 // AdvSIMD registers/instructions as long as the copies to/from the AdvSIMD
2365 // registers are cheap.
2366 // FMOV D0, X0 // copy 64-bit int to vector, high bits zero'd
2367 // CNT V0.8B, V0.8B // 8xbyte pop-counts
2368 // ADDV B0, V0.8B // sum 8xbyte pop-counts
2369 // UMOV X0, V0.B[0] // copy byte result back to integer reg
2370 //
2371 // For 128 bit vector popcounts, we lower to the following sequence:
2372 // cnt.16b v0, v0 // v8s16, v4s32, v2s64
2373 // uaddlp.8h v0, v0 // v8s16, v4s32, v2s64
2374 // uaddlp.4s v0, v0 // v4s32, v2s64
2375 // uaddlp.2d v0, v0 // v2s64
2376 //
2377 // For 64 bit vector popcounts, we lower to the following sequence:
2378 // cnt.8b v0, v0 // v4s16, v2s32
2379 // uaddlp.4h v0, v0 // v4s16, v2s32
2380 // uaddlp.2s v0, v0 // v2s32
2381
2382 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
2383 Register Dst = MI.getOperand(0).getReg();
2384 Register Val = MI.getOperand(1).getReg();
2385 LLT Ty = MRI.getType(Val);
2386
2387 LLT i64 = LLT::integer(64);
2388 LLT i32 = LLT::integer(32);
2389 LLT i16 = LLT::integer(16);
2390 LLT i8 = LLT::integer(8);
2391 unsigned Size = Ty.getSizeInBits();
2392
2393 assert(Ty == MRI.getType(Dst) &&
2394 "Expected src and dst to have the same type!");
2395
2396 if (ST->hasCSSC() && Ty.isScalar() && Size == 128) {
2397
2398 auto Split = MIRBuilder.buildUnmerge(i64, Val);
2399 auto CTPOP1 = MIRBuilder.buildCTPOP(i64, Split->getOperand(0));
2400 auto CTPOP2 = MIRBuilder.buildCTPOP(i64, Split->getOperand(1));
2401 auto Add = MIRBuilder.buildAdd(i64, CTPOP1, CTPOP2);
2402
2403 MIRBuilder.buildZExt(Dst, Add);
2404 MI.eraseFromParent();
2405 return true;
2406 }
2407
2408 if (!ST->hasNEON() ||
2409 MI.getMF()->getFunction().hasFnAttribute(Attribute::NoImplicitFloat)) {
2410 // Use generic lowering when custom lowering is not possible.
2411 return Ty.isScalar() && (Size == 32 || Size == 64) &&
2412 Helper.lowerBitCount(MI) ==
2414 }
2415
2416 // Pre-conditioning: widen Val up to the nearest vector type.
2417 // s32,s64,v4s16,v2s32 -> v8i8
2418 // v8s16,v4s32,v2s64 -> v16i8
2419 LLT VTy = Size == 128 ? LLT::fixed_vector(16, i8) : LLT::fixed_vector(8, i8);
2420 if (Ty.isScalar()) {
2421 assert((Size == 32 || Size == 64 || Size == 128) && "Expected only 32, 64, or 128 bit scalars!");
2422 if (Size == 32) {
2423 Val = MIRBuilder.buildZExt(i64, Val).getReg(0);
2424 }
2425 }
2426 Val = MIRBuilder.buildBitcast(VTy, Val).getReg(0);
2427
2428 // Count bits in each byte-sized lane.
2429 auto CTPOP = MIRBuilder.buildCTPOP(VTy, Val);
2430
2431 // Sum across lanes.
2432 if (ST->hasDotProd() && Ty.isVector() && Ty.getNumElements() >= 2 &&
2433 Ty.getScalarSizeInBits() != 16) {
2434 LLT Dt = Ty == LLT::fixed_vector(2, i64) ? LLT::fixed_vector(4, i32) : Ty;
2435 auto Zeros = MIRBuilder.buildConstant(Dt, 0);
2436 auto Ones = MIRBuilder.buildConstant(VTy, 1);
2437 MachineInstrBuilder Sum;
2438
2439 if (Ty == LLT::fixed_vector(2, i64)) {
2440 auto UDOT =
2441 MIRBuilder.buildInstr(AArch64::G_UDOT, {Dt}, {Zeros, Ones, CTPOP});
2442 Sum = MIRBuilder.buildInstr(AArch64::G_UADDLP, {Ty}, {UDOT});
2443 } else if (Ty == LLT::fixed_vector(4, i32)) {
2444 Sum = MIRBuilder.buildInstr(AArch64::G_UDOT, {Dt}, {Zeros, Ones, CTPOP});
2445 } else if (Ty == LLT::fixed_vector(2, i32)) {
2446 Sum = MIRBuilder.buildInstr(AArch64::G_UDOT, {Dt}, {Zeros, Ones, CTPOP});
2447 } else {
2448 llvm_unreachable("unexpected vector shape");
2449 }
2450
2451 Sum->getOperand(0).setReg(Dst);
2452 MI.eraseFromParent();
2453 return true;
2454 }
2455
2456 Register HSum = CTPOP.getReg(0);
2457 unsigned Opc;
2458 SmallVector<LLT> HAddTys;
2459 if (Ty.isScalar()) {
2460 Opc = Intrinsic::aarch64_neon_uaddlv;
2461 HAddTys.push_back(i32);
2462 } else if (Ty == LLT::fixed_vector(8, i16)) {
2463 Opc = Intrinsic::aarch64_neon_uaddlp;
2464 HAddTys.push_back(LLT::fixed_vector(8, i16));
2465 } else if (Ty == LLT::fixed_vector(4, i32)) {
2466 Opc = Intrinsic::aarch64_neon_uaddlp;
2467 HAddTys.push_back(LLT::fixed_vector(8, i16));
2468 HAddTys.push_back(LLT::fixed_vector(4, i32));
2469 } else if (Ty == LLT::fixed_vector(2, i64)) {
2470 Opc = Intrinsic::aarch64_neon_uaddlp;
2471 HAddTys.push_back(LLT::fixed_vector(8, i16));
2472 HAddTys.push_back(LLT::fixed_vector(4, i32));
2473 HAddTys.push_back(LLT::fixed_vector(2, i64));
2474 } else if (Ty == LLT::fixed_vector(4, i16)) {
2475 Opc = Intrinsic::aarch64_neon_uaddlp;
2476 HAddTys.push_back(LLT::fixed_vector(4, i16));
2477 } else if (Ty == LLT::fixed_vector(2, i32)) {
2478 Opc = Intrinsic::aarch64_neon_uaddlp;
2479 HAddTys.push_back(LLT::fixed_vector(4, i16));
2480 HAddTys.push_back(LLT::fixed_vector(2, i32));
2481 } else
2482 llvm_unreachable("unexpected vector shape");
2484 for (LLT HTy : HAddTys) {
2485 UADD = MIRBuilder.buildIntrinsic(Opc, {HTy}).addUse(HSum);
2486 HSum = UADD.getReg(0);
2487 }
2488
2489 // Post-conditioning.
2490 if (Ty.isScalar() && (Size == 64 || Size == 128))
2491 MIRBuilder.buildZExt(Dst, UADD);
2492 else
2493 UADD->getOperand(0).setReg(Dst);
2494 MI.eraseFromParent();
2495 return true;
2496}
2497
2498bool AArch64LegalizerInfo::legalizeAtomicCmpxchg128(
2499 MachineInstr &MI, MachineRegisterInfo &MRI, LegalizerHelper &Helper) const {
2500 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
2501 LLT i64 = LLT::integer(64);
2502 auto Addr = MI.getOperand(1).getReg();
2503 auto DesiredI = MIRBuilder.buildUnmerge({i64, i64}, MI.getOperand(2));
2504 auto NewI = MIRBuilder.buildUnmerge({i64, i64}, MI.getOperand(3));
2505 auto DstLo = MRI.createGenericVirtualRegister(i64);
2506 auto DstHi = MRI.createGenericVirtualRegister(i64);
2507
2508 MachineInstrBuilder CAS;
2509 if (ST->hasLSE()) {
2510 // We have 128-bit CASP instructions taking XSeqPair registers, which are
2511 // s128. We need the merge/unmerge to bracket the expansion and pair up with
2512 // the rest of the MIR so we must reassemble the extracted registers into a
2513 // 128-bit known-regclass one with code like this:
2514 //
2515 // %in1 = REG_SEQUENCE Lo, Hi ; One for each input
2516 // %out = CASP %in1, ...
2517 // %OldLo = G_EXTRACT %out, 0
2518 // %OldHi = G_EXTRACT %out, 64
2519 auto Ordering = (*MI.memoperands_begin())->getMergedOrdering();
2520 unsigned Opcode;
2521 switch (Ordering) {
2523 Opcode = AArch64::CASPAX;
2524 break;
2526 Opcode = AArch64::CASPLX;
2527 break;
2530 Opcode = AArch64::CASPALX;
2531 break;
2532 default:
2533 Opcode = AArch64::CASPX;
2534 break;
2535 }
2536
2537 LLT s128 = LLT::integer(128);
2538 auto CASDst = MRI.createGenericVirtualRegister(s128);
2539 auto CASDesired = MRI.createGenericVirtualRegister(s128);
2540 auto CASNew = MRI.createGenericVirtualRegister(s128);
2541 MIRBuilder.buildInstr(TargetOpcode::REG_SEQUENCE, {CASDesired}, {})
2542 .addUse(DesiredI->getOperand(0).getReg())
2543 .addImm(AArch64::sube64)
2544 .addUse(DesiredI->getOperand(1).getReg())
2545 .addImm(AArch64::subo64);
2546 MIRBuilder.buildInstr(TargetOpcode::REG_SEQUENCE, {CASNew}, {})
2547 .addUse(NewI->getOperand(0).getReg())
2548 .addImm(AArch64::sube64)
2549 .addUse(NewI->getOperand(1).getReg())
2550 .addImm(AArch64::subo64);
2551
2552 CAS = MIRBuilder.buildInstr(Opcode, {CASDst}, {CASDesired, CASNew, Addr});
2553
2554 MIRBuilder.buildExtract({DstLo}, {CASDst}, 0);
2555 MIRBuilder.buildExtract({DstHi}, {CASDst}, 64);
2556 } else {
2557 // The -O0 CMP_SWAP_128 is friendlier to generate code for because LDXP/STXP
2558 // can take arbitrary registers so it just has the normal GPR64 operands the
2559 // rest of AArch64 is expecting.
2560 auto Ordering = (*MI.memoperands_begin())->getMergedOrdering();
2561 unsigned Opcode;
2562 switch (Ordering) {
2564 Opcode = AArch64::CMP_SWAP_128_ACQUIRE;
2565 break;
2567 Opcode = AArch64::CMP_SWAP_128_RELEASE;
2568 break;
2571 Opcode = AArch64::CMP_SWAP_128;
2572 break;
2573 default:
2574 Opcode = AArch64::CMP_SWAP_128_MONOTONIC;
2575 break;
2576 }
2577
2578 auto Scratch = MRI.createVirtualRegister(&AArch64::GPR64RegClass);
2579 CAS = MIRBuilder.buildInstr(Opcode, {DstLo, DstHi, Scratch},
2580 {Addr, DesiredI->getOperand(0),
2581 DesiredI->getOperand(1), NewI->getOperand(0),
2582 NewI->getOperand(1)});
2583 }
2584
2585 CAS.cloneMemRefs(MI);
2586 constrainSelectedInstRegOperands(*CAS, *ST->getInstrInfo(),
2587 *MRI.getTargetRegisterInfo(),
2588 *ST->getRegBankInfo());
2589
2590 MIRBuilder.buildMergeLikeInstr(MI.getOperand(0), {DstLo, DstHi});
2591 MI.eraseFromParent();
2592 return true;
2593}
2594
2595bool AArch64LegalizerInfo::legalizeCTTZ(MachineInstr &MI,
2596 LegalizerHelper &Helper) const {
2597 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
2598 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
2599 LLT Ty = MRI.getType(MI.getOperand(1).getReg());
2600 auto BitReverse = MIRBuilder.buildBitReverse(Ty, MI.getOperand(1));
2601 MIRBuilder.buildCTLZ(MI.getOperand(0).getReg(), BitReverse);
2602 MI.eraseFromParent();
2603 return true;
2604}
2605
2606bool AArch64LegalizerInfo::legalizeMemOps(MachineInstr &MI,
2607 LegalizerHelper &Helper) const {
2608 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
2609
2610 // Tagged version MOPSMemorySetTagged is legalised in legalizeIntrinsic
2611 if (MI.getOpcode() == TargetOpcode::G_MEMSET ||
2612 MI.getOpcode() == TargetOpcode::G_MEMSET_INLINE) {
2613 // Anyext the value being set to 64 bit (only the bottom 8 bits are read by
2614 // the instruction).
2615 auto &Value = MI.getOperand(1);
2616 Register ExtValueReg =
2617 MIRBuilder.buildAnyExt(LLT::integer(64), Value).getReg(0);
2618 Value.setReg(ExtValueReg);
2619 return true;
2620 }
2621
2622 return false;
2623}
2624
2625bool AArch64LegalizerInfo::legalizeExtractVectorElt(
2626 MachineInstr &MI, MachineRegisterInfo &MRI, LegalizerHelper &Helper) const {
2627 const GExtractVectorElement *Element = cast<GExtractVectorElement>(&MI);
2628 auto VRegAndVal =
2630 if (VRegAndVal)
2631 return true;
2632 LLT VecTy = MRI.getType(Element->getVectorReg());
2633 if (VecTy.isScalableVector())
2634 return true;
2635 return Helper.lowerExtractInsertVectorElt(MI) !=
2637}
2638
2639bool AArch64LegalizerInfo::legalizeDynStackAlloc(
2640 MachineInstr &MI, LegalizerHelper &Helper) const {
2641 MachineFunction &MF = *MI.getParent()->getParent();
2642 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
2643 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
2644
2645 // If stack probing is not enabled for this function, use the default
2646 // lowering.
2647 if (!MF.getFunction().hasFnAttribute("probe-stack") ||
2648 MF.getFunction().getFnAttribute("probe-stack").getValueAsString() !=
2649 "inline-asm") {
2650 Helper.lowerDynStackAlloc(MI);
2651 return true;
2652 }
2653
2654 Register Dst = MI.getOperand(0).getReg();
2655 Register AllocSize = MI.getOperand(1).getReg();
2656 Align Alignment = assumeAligned(MI.getOperand(2).getImm());
2657
2658 assert(MRI.getType(Dst) == LLT::pointer(0, 64) &&
2659 "Unexpected type for dynamic alloca");
2660 assert(MRI.getType(AllocSize) == LLT::scalar(64) &&
2661 "Unexpected type for dynamic alloca");
2662
2663 LLT PtrTy = MRI.getType(Dst);
2664 Register SPReg =
2666 Register SPTmp =
2667 Helper.getDynStackAllocTargetPtr(SPReg, AllocSize, Alignment, PtrTy);
2668 auto NewMI =
2669 MIRBuilder.buildInstr(AArch64::PROBED_STACKALLOC_DYN, {}, {SPTmp});
2670 MRI.setRegClass(NewMI.getReg(0), &AArch64::GPR64commonRegClass);
2671 MIRBuilder.setInsertPt(*NewMI->getParent(), NewMI);
2672 MIRBuilder.buildCopy(Dst, SPTmp);
2673
2674 MI.eraseFromParent();
2675 return true;
2676}
2677
2678bool AArch64LegalizerInfo::legalizePrefetch(MachineInstr &MI,
2679 LegalizerHelper &Helper) const {
2680 MachineIRBuilder &MIB = Helper.MIRBuilder;
2681 auto &AddrVal = MI.getOperand(0);
2682
2683 int64_t IsWrite = MI.getOperand(1).getImm();
2684 int64_t Locality = MI.getOperand(2).getImm();
2685 int64_t IsData = MI.getOperand(3).getImm();
2686
2687 bool IsStream = Locality == 0;
2688 if (Locality != 0) {
2689 assert(Locality <= 3 && "Prefetch locality out-of-range");
2690 // The locality degree is the opposite of the cache speed.
2691 // Put the number the other way around.
2692 // The encoding starts at 0 for level 1
2693 Locality = 3 - Locality;
2694 }
2695
2696 unsigned PrfOp = (IsWrite << 4) | (!IsData << 3) | (Locality << 1) | IsStream;
2697
2698 MIB.buildInstr(AArch64::G_AARCH64_PREFETCH).addImm(PrfOp).add(AddrVal);
2699 MI.eraseFromParent();
2700 return true;
2701}
2702
2703bool AArch64LegalizerInfo::legalizeConcatVectors(
2705 MachineIRBuilder &MIRBuilder) const {
2706 // Widen sub-byte element vectors to byte-sized elements before concatenating.
2707 // This is analogous to SDAG's integer type promotion for sub-byte types.
2709 Register DstReg = Concat.getReg(0);
2710 LLT DstTy = MRI.getType(DstReg);
2711 assert(DstTy.getScalarSizeInBits() < 8 && "Expected dst ty to be < 8b");
2712
2713 unsigned WideEltSize =
2714 std::max(8u, (unsigned)PowerOf2Ceil(DstTy.getScalarSizeInBits()));
2715 LLT SrcTy = MRI.getType(Concat.getSourceReg(0));
2716 LLT WideSrcTy = SrcTy.changeElementSize(WideEltSize);
2717 LLT WideDstTy = DstTy.changeElementSize(WideEltSize);
2718
2719 SmallVector<Register> WideSrcs;
2720 for (unsigned I = 0; I < Concat.getNumSources(); ++I) {
2721 auto Wide = MIRBuilder.buildAnyExt(WideSrcTy, Concat.getSourceReg(I));
2722 WideSrcs.push_back(Wide.getReg(0));
2723 }
2724
2725 auto WideConcat = MIRBuilder.buildConcatVectors(WideDstTy, WideSrcs);
2726 MIRBuilder.buildTrunc(DstReg, WideConcat);
2727 MI.eraseFromParent();
2728 return true;
2729}
2730
2731bool AArch64LegalizerInfo::legalizeFptrunc(MachineInstr &MI,
2732 MachineIRBuilder &MIRBuilder,
2733 MachineRegisterInfo &MRI) const {
2734 auto [Dst, DstTy, Src, SrcTy] = MI.getFirst2RegLLTs();
2735
2736 // This function legalizes f64 -> bf16 and f64 -> f16 truncations via f64 ->
2737 // f32 G_FPTRUNC_ODD and f32 -> [b]f16 G_FPTRUNC, which apparently avoids the
2738 // usual double-rounding issue that could be present from using twin
2739 // G_FPTRUNC.
2740
2741 if (DstTy.isBFloat16() && SrcTy.isFloat64()) {
2742 auto Mid = MIRBuilder.buildInstr(AArch64::G_FPTRUNC_ODD, {LLT::float32()},
2743 {Src}, MI.getFlags());
2744 MIRBuilder.buildInstr(AArch64::G_FPTRUNC, {Dst}, {Mid}, MI.getFlags());
2745 MI.eraseFromParent();
2746 return true;
2747 }
2748
2749 assert(SrcTy.isFixedVector() && isPowerOf2_32(SrcTy.getNumElements()) &&
2750 "Expected a power of 2 elements");
2751
2752 // We must mutate types here as FPTrunc may be used on a IEEE floating point
2753 // or a brainfloat.
2754 LLT v2s16 = DstTy.changeElementCount(2);
2755 LLT v4s16 = DstTy.changeElementCount(4);
2756 LLT v2s32 = SrcTy.changeElementCount(2).changeElementSize(32);
2757 LLT v4s32 = SrcTy.changeElementCount(4).changeElementSize(32);
2758 LLT v2s64 = SrcTy.changeElementCount(2);
2759
2760 SmallVector<Register> RegsToUnmergeTo;
2761 SmallVector<Register> TruncOddDstRegs;
2762 SmallVector<Register> RegsToMerge;
2763
2764 unsigned ElemCount = SrcTy.getNumElements();
2765
2766 // Find the biggest size chunks we can work with
2767 int StepSize = ElemCount % 4 ? 2 : 4;
2768
2769 // If we have a power of 2 greater than 2, we need to first unmerge into
2770 // enough pieces
2771 if (ElemCount <= 2)
2772 RegsToUnmergeTo.push_back(Src);
2773 else {
2774 for (unsigned i = 0; i < ElemCount / 2; ++i)
2775 RegsToUnmergeTo.push_back(MRI.createGenericVirtualRegister(v2s64));
2776
2777 MIRBuilder.buildUnmerge(RegsToUnmergeTo, Src);
2778 }
2779
2780 // Create all of the round-to-odd instructions and store them
2781 for (auto SrcReg : RegsToUnmergeTo) {
2782 Register Mid = MIRBuilder
2783 .buildInstr(AArch64::G_FPTRUNC_ODD, {v2s32}, {SrcReg},
2784 MI.getFlags())
2785 .getReg(0);
2786 TruncOddDstRegs.push_back(Mid);
2787 }
2788
2789 // Truncate 4s32 to 4s16 if we can to reduce instruction count, otherwise
2790 // truncate 2s32 to 2s16.
2791 unsigned Index = 0;
2792 for (unsigned LoopIter = 0; LoopIter < ElemCount / StepSize; ++LoopIter) {
2793 if (StepSize == 4) {
2794 Register ConcatDst =
2795 MIRBuilder
2797 {v4s32}, {TruncOddDstRegs[Index++], TruncOddDstRegs[Index++]})
2798 .getReg(0);
2799
2800 RegsToMerge.push_back(
2801 MIRBuilder.buildFPTrunc(v4s16, ConcatDst, MI.getFlags()).getReg(0));
2802 } else {
2803 RegsToMerge.push_back(
2804 MIRBuilder
2805 .buildFPTrunc(v2s16, TruncOddDstRegs[Index++], MI.getFlags())
2806 .getReg(0));
2807 }
2808 }
2809
2810 // If there is only one register, replace the destination
2811 if (RegsToMerge.size() == 1) {
2812 MRI.replaceRegWith(Dst, RegsToMerge.pop_back_val());
2813 MI.eraseFromParent();
2814 return true;
2815 }
2816
2817 // Merge the rest of the instructions & replace the register
2818 Register Fin = MIRBuilder.buildMergeLikeInstr(DstTy, RegsToMerge).getReg(0);
2819 MRI.replaceRegWith(Dst, Fin);
2820 MI.eraseFromParent();
2821 return true;
2822}
2823
2824bool AArch64LegalizerInfo::legalizeGetRounding(MachineInstr &MI,
2825 MachineIRBuilder &MIRBuilder,
2827 LegalizerHelper &Helper) const {
2828 const LLT I32 = LLT::integer(32);
2829 const LLT I64 = LLT::integer(64);
2830
2831 Register Dst = MI.getOperand(0).getReg();
2832 Register FPCR64 = MRI.createGenericVirtualRegister(I64);
2833 MachineInstrBuilder GetFPCR =
2834 MIRBuilder.buildIntrinsic(Intrinsic::aarch64_get_fpcr, ArrayRef{FPCR64});
2835
2836 // AArch64 rounding mode value to FLT_ROUNDS mapping is 0->1, 1->2, 2->3,
2837 // 3->0, so we add one to the FPCR bits for the rounding mode.
2838 // Instead of shifting and then adding as `((FPCR >> 22) + 1) & 0b11` which
2839 // generates 3 instructions, we increment the rounding mode with
2840 // `(FPCR + (1 << 22))` and extract the bits. The shift and addition is done
2841 // in one instruction as `add .., .., #1024, lsl #12`, so overall we generate
2842 // one less instruction.
2843 auto FPCR32 = MIRBuilder.buildTrunc(I32, GetFPCR);
2844 auto One = MIRBuilder.buildConstant(I32, 1U << 22);
2845 auto Added = MIRBuilder.buildAdd(I32, FPCR32, One);
2846 auto LSB = MIRBuilder.buildConstant(I32, 22);
2847 auto Width = MIRBuilder.buildConstant(I32, 2);
2848 MIRBuilder.buildInstr(TargetOpcode::G_UBFX, {Dst}, {Added, LSB, Width});
2849
2850 MI.eraseFromParent();
2851 return true;
2852}
2853
2854bool AArch64LegalizerInfo::legalizeSetRounding(MachineInstr &MI,
2855 MachineIRBuilder &MIRBuilder,
2857 LegalizerHelper &Helper) const {
2858 const LLT I32 = LLT::integer(32);
2859 const LLT I64 = LLT::integer(64);
2860
2861 // AArch64 rounding mode value to FLT_ROUNDS mapping is 0->1, 1->2, 2->3,
2862 // 3->0, so calculate the new value of FPCR[23:22] as `((arg - 1) & 3) << 22`.
2863 Register RM = MI.getOperand(0).getReg();
2864 auto One = MIRBuilder.buildConstant(I32, 1);
2865 auto Subtracted = MIRBuilder.buildSub(I32, RM, One);
2866 auto Mask = MIRBuilder.buildConstant(I32, 0b11);
2867 auto Masked = MIRBuilder.buildAnd(I32, Subtracted, Mask);
2868 auto ShiftAmount = MIRBuilder.buildConstant(I32, 22);
2869 auto Shifted = MIRBuilder.buildShl(I32, Masked, ShiftAmount);
2870
2871 // Get current value of FPCR.
2872 MachineInstrBuilder GetFPCR =
2873 MIRBuilder.buildIntrinsic(Intrinsic::aarch64_get_fpcr, {I64});
2874
2875 // (FPCR & ~Mask) | Shifted
2876 auto FPCRMask = MIRBuilder.buildConstant(I64, ~((int64_t)0b11 << 22));
2877 auto FPCRMasked = MIRBuilder.buildAnd(I64, GetFPCR, FPCRMask);
2878 auto ShiftedS64 = MIRBuilder.buildZExt(I64, Shifted);
2879 auto FPCRUpdated = MIRBuilder.buildOr(I64, FPCRMasked, ShiftedS64);
2880
2881 // Write new FPCR.
2882 MIRBuilder.buildIntrinsic(Intrinsic::aarch64_set_fpcr, ArrayRef<Register>())
2883 .addUse(FPCRUpdated.getReg(0));
2884
2885 MI.eraseFromParent();
2886
2887 return true;
2888}
static void matchLDPSTPAddrMode(Register Root, Register &Base, int &Offset, MachineRegisterInfo &MRI)
This file declares the targeting of the Machinelegalizer class for AArch64.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned uint64_t
static Error unsupported(const char *Str, const Triple &T)
Definition MachO.cpp:79
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
IRTranslator LLVM IR MI
Interface for Targets to specify which operations they can successfully select and how the others sho...
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineIRBuilder class.
Promote Memory to Register
Definition Mem2Reg.cpp:110
static MCRegister getReg(const MCDisassembler *D, unsigned RC, unsigned RegNo)
ppc ctr loops verify
if(PassOpts->AAPipeline)
static constexpr MCPhysReg SPReg
This file contains some templates that are useful if you are working with the STL at all.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static constexpr int Concat[]
bool legalizeCustom(LegalizerHelper &Helper, MachineInstr &MI, LostDebugLocObserver &LocObserver) const override
Called for instructions with the Custom LegalizationAction.
bool legalizeIntrinsic(LegalizerHelper &Helper, MachineInstr &MI) const override
AArch64LegalizerInfo(const AArch64Subtarget &ST)
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
LLVM_ABI APInt urem(const APInt &RHS) const
Unsigned remainder operation.
Definition APInt.cpp:1695
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
Predicate
This enumeration lists the possible predicates for CmpInst subclasses.
Definition InstrTypes.h:740
@ ICMP_NE
not equal
Definition InstrTypes.h:762
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:769
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
Abstract class that contains various methods for clients to notify about changes.
virtual void changingInstr(MachineInstr &MI)=0
This instruction is about to be mutated in some way.
virtual void changedInstr(MachineInstr &MI)=0
This instruction was mutated in some way.
static constexpr LLT float64()
Get a 64-bit IEEE double value.
LLT changeElementCount(ElementCount EC) const
Return a vector or scalar with the same element type and the new element count.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr bool isScalar() const
static constexpr LLT scalable_vector(unsigned MinNumElements, unsigned ScalarSizeInBits)
Get a low-level scalable vector of some number of elements and element width.
static constexpr LLT vector(ElementCount EC, unsigned ScalarSizeInBits)
Get a low-level vector of some number of elements and element width.
LLT getScalarType() const
constexpr bool isPointerVector() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr uint16_t getNumElements() const
Returns the number of elements in a vector LLT.
static constexpr LLT float128()
Get a 128-bit IEEE quad value.
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr ElementCount getElementCount() const
LLT divide(int Factor) const
Return a type that is Factor times smaller.
static constexpr LLT float16()
Get a 16-bit IEEE half value.
constexpr unsigned getAddressSpace() const
static constexpr LLT fixed_vector(unsigned NumElements, unsigned ScalarSizeInBits)
Get a low-level fixed-width vector of some number of elements and element width.
constexpr bool isFixedVector() const
Returns true if the LLT is a fixed vector.
static LLT integer(unsigned SizeInBits)
static constexpr LLT bfloat16()
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
static constexpr LLT float32()
Get a 32-bit IEEE float value.
bool isFloat64() const
LLT changeElementSize(unsigned NewEltSize) const
If this type is a vector, return a vector with the same number of elements but the new element size.
LegalizeRuleSet & minScalar(unsigned TypeIdx, const LLT Ty)
Ensure the scalar is at least as wide as Ty.
LegalizeRuleSet & widenScalarOrEltToNextPow2OrMinSize(unsigned TypeIdx, unsigned MinSize=0)
Widen the scalar or vector element type to the next power of two that is at least MinSize.
LegalizeRuleSet & legalFor(std::initializer_list< LLT > Types)
The instruction is legal when type index 0 is any type in the given list.
LegalizeRuleSet & maxScalarEltSameAsIf(LegalityPredicate Predicate, unsigned TypeIdx, unsigned SmallTypeIdx)
Conditionally narrow the scalar or elt to match the size of another.
LegalizeRuleSet & unsupported()
The instruction is unsupported.
LegalizeRuleSet & scalarSameSizeAs(unsigned TypeIdx, unsigned SameSizeIdx)
Change the type TypeIdx to have the same scalar size as type SameSizeIdx.
LegalizeRuleSet & bitcastIf(LegalityPredicate Predicate, LegalizeMutation Mutation)
The specified type index is coerced if predicate is true.
LegalizeRuleSet & libcallFor(std::initializer_list< LLT > Types)
LegalizeRuleSet & minScalarOrElt(unsigned TypeIdx, const LLT Ty)
Ensure the scalar or element is at least as wide as Ty.
LegalizeRuleSet & clampMaxNumElements(unsigned TypeIdx, const LLT EltTy, unsigned MaxElements)
Limit the number of elements in EltTy vectors to at most MaxElements.
LegalizeRuleSet & clampMinNumElements(unsigned TypeIdx, const LLT EltTy, unsigned MinElements)
Limit the number of elements in EltTy vectors to at least MinElements.
LegalizeRuleSet & widenVectorEltsToVectorMinSize(unsigned TypeIdx, unsigned VectorSize)
Ensure the vector size is at least as wide as VectorSize by promoting the element.
LegalizeRuleSet & lowerIfMemSizeNotPow2()
Lower a memory operation if the memory size, rounded to bytes, is not a power of 2.
LegalizeRuleSet & minScalarEltSameAsIf(LegalityPredicate Predicate, unsigned TypeIdx, unsigned LargeTypeIdx)
Conditionally widen the scalar or elt to match the size of another.
LegalizeRuleSet & customForCartesianProduct(std::initializer_list< LLT > Types)
LegalizeRuleSet & lowerIfMemSizeNotByteSizePow2()
Lower a memory operation if the memory access size is not a round power of 2 byte size.
LegalizeRuleSet & moreElementsToNextPow2(unsigned TypeIdx)
Add more elements to the vector to reach the next power of two.
LegalizeRuleSet & narrowScalarIf(LegalityPredicate Predicate, LegalizeMutation Mutation)
Narrow the scalar to the one selected by the mutation if the predicate is true.
LegalizeRuleSet & lower()
The instruction is lowered.
LegalizeRuleSet & moreElementsIf(LegalityPredicate Predicate, LegalizeMutation Mutation)
Add more elements to reach the type selected by the mutation if the predicate is true.
LegalizeRuleSet & lowerFor(std::initializer_list< LLT > Types)
The instruction is lowered when type index 0 is any type in the given list.
LegalizeRuleSet & scalarizeIf(LegalityPredicate Predicate, unsigned TypeIdx)
LegalizeRuleSet & lowerIf(LegalityPredicate Predicate)
The instruction is lowered if predicate is true.
LegalizeRuleSet & clampScalar(unsigned TypeIdx, const LLT MinTy, const LLT MaxTy)
Limit the range of scalar sizes to MinTy and MaxTy.
LegalizeRuleSet & custom()
Unconditionally custom lower.
LegalizeRuleSet & minScalarSameAs(unsigned TypeIdx, unsigned LargeTypeIdx)
Widen the scalar to match the size of another.
LegalizeRuleSet & unsupportedIf(LegalityPredicate Predicate)
LegalizeRuleSet & minScalarOrEltIf(LegalityPredicate Predicate, unsigned TypeIdx, const LLT Ty)
Ensure the scalar or element is at least as wide as Ty.
LegalizeRuleSet & widenScalarIf(LegalityPredicate Predicate, LegalizeMutation Mutation)
Widen the scalar to the one selected by the mutation if the predicate is true.
LegalizeRuleSet & alwaysLegal()
LegalizeRuleSet & clampNumElements(unsigned TypeIdx, const LLT MinTy, const LLT MaxTy)
Limit the number of elements for the given vectors to at least MinTy's number of elements and at most...
LegalizeRuleSet & maxScalarIf(LegalityPredicate Predicate, unsigned TypeIdx, const LLT Ty)
Conditionally limit the maximum size of the scalar.
LegalizeRuleSet & customIf(LegalityPredicate Predicate)
LegalizeRuleSet & widenScalarToNextPow2(unsigned TypeIdx, unsigned MinSize=0)
Widen the scalar to the next power of two that is at least MinSize.
LegalizeRuleSet & scalarize(unsigned TypeIdx)
LegalizeRuleSet & legalForCartesianProduct(std::initializer_list< LLT > Types)
The instruction is legal when type indexes 0 and 1 are both in the given list.
LegalizeRuleSet & legalForTypesWithMemDesc(std::initializer_list< LegalityPredicates::TypePairAndMemDesc > TypesAndMemDesc)
The instruction is legal when type indexes 0 and 1 along with the memory size and minimum alignment i...
LegalizeRuleSet & legalIf(LegalityPredicate Predicate)
The instruction is legal if predicate is true.
LLVM_ABI LegalizeResult lowerDynStackAlloc(MachineInstr &MI)
LLVM_ABI LegalizeResult lowerBitCount(MachineInstr &MI)
LLVM_ABI LegalizeResult lowerExtractInsertVectorElt(MachineInstr &MI)
Lower a vector extract or insert by writing the vector to a stack temporary and reloading the element...
LLVM_ABI LegalizeResult lowerAbsToCNeg(MachineInstr &MI)
const TargetLowering & getTargetLowering() const
LLVM_ABI LegalizeResult lowerFunnelShiftAsShifts(MachineInstr &MI)
LLVM_ABI MachineInstrBuilder createStackStoreLoad(const DstOp &Res, const SrcOp &Val)
Create a store of Val to a stack temporary and return a load as the same type as Res.
@ Legalized
Instruction has been legalized and the MachineFunction changed.
@ UnableToLegalize
Some kind of error has occurred and we could not legalize this instruction.
GISelChangeObserver & Observer
To keep track of changes made by the LegalizerHelper.
LLVM_ABI Register getDynStackAllocTargetPtr(Register SPReg, Register AllocSize, Align Alignment, LLT PtrTy)
MachineIRBuilder & MIRBuilder
Expose MIRBuilder so clients can set their own RecordInsertInstruction functions.
LegalizeRuleSet & getActionDefinitionsBuilder(unsigned Opcode)
Get the action definition builder for the given opcode.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
Helper class to build MachineInstr.
void setInsertPt(MachineBasicBlock &MBB, MachineBasicBlock::iterator II)
Set the insertion point before the specified position.
MachineInstrBuilder buildAdd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ADD Op0, Op1.
MachineInstrBuilder buildNot(const DstOp &Dst, const SrcOp &Src0)
Build and insert a bitwise not, NegOne = G_CONSTANT -1 Res = G_OR Op0, NegOne.
MachineInstrBuilder buildUnmerge(ArrayRef< LLT > Res, const SrcOp &Op)
Build and insert Res0, ... = G_UNMERGE_VALUES Op.
MachineInstrBuilder buildExtract(const DstOp &Res, const SrcOp &Src, uint64_t Index)
Build and insert Res0, ... = G_EXTRACT Src, Idx0.
MachineInstrBuilder buildAnd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1)
Build and insert Res = G_AND Op0, Op1.
MachineInstrBuilder buildICmp(CmpInst::Predicate Pred, const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert a Res = G_ICMP Pred, Op0, Op1.
MachineBasicBlock::iterator getInsertPt()
Current insertion point for new instructions.
MachineInstrBuilder buildZExt(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ZEXT Op.
MachineInstrBuilder buildConcatVectors(const DstOp &Res, ArrayRef< Register > Ops)
Build and insert Res = G_CONCAT_VECTORS Op0, ...
MachineInstrBuilder buildSub(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_SUB Op0, Op1.
MachineInstrBuilder buildIntrinsic(Intrinsic::ID ID, ArrayRef< Register > Res, bool HasSideEffects, bool isConvergent)
Build and insert a G_INTRINSIC instruction.
MachineInstrBuilder buildCTLZ(const DstOp &Dst, const SrcOp &Src0)
Build and insert Res = G_CTLZ Op0, Src0.
MachineInstrBuilder buildMergeLikeInstr(const DstOp &Res, ArrayRef< Register > Ops)
Build and insert Res = G_MERGE_VALUES Op0, ... or Res = G_BUILD_VECTOR Op0, ... or Res = G_CONCAT_VEC...
MachineInstrBuilder buildLoad(const DstOp &Res, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert Res = G_LOAD Addr, MMO.
MachineInstrBuilder buildPtrAdd(const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_PTR_ADD Op0, Op1.
MachineInstrBuilder buildBitReverse(const DstOp &Dst, const SrcOp &Src)
Build and insert Dst = G_BITREVERSE Src.
MachineInstrBuilder buildShl(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
MachineInstrBuilder buildStore(const SrcOp &Val, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert G_STORE Val, Addr, MMO.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
MachineInstrBuilder buildCTPOP(const DstOp &Dst, const SrcOp &Src0)
Build and insert Res = G_CTPOP Op0, Src0.
MachineFunction & getMF()
Getter for the function we currently build.
MachineInstrBuilder buildExtOrTrunc(unsigned ExtOpc, const DstOp &Res, const SrcOp &Op)
Build and insert Res = ExtOpc, Res = G_TRUNC Op, or Res = COPY Op depending on the differing sizes of...
MachineInstrBuilder buildTrunc(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_TRUNC Op.
const MachineBasicBlock & getMBB() const
Getter for the basic block we currently build.
MachineInstrBuilder buildAnyExt(const DstOp &Res, const SrcOp &Op)
Build and insert Res = G_ANYEXT Op0.
MachineInstrBuilder buildBitcast(const DstOp &Dst, const SrcOp &Src)
Build and insert Dst = G_BITCAST Src.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineInstrBuilder buildFPTrunc(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_FPTRUNC Op.
MachineInstrBuilder buildOr(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_OR Op0, Op1.
MachineInstrBuilder buildCopy(const DstOp &Res, const SrcOp &Op)
Build and insert Res = COPY Op.
MachineInstrBuilder buildMaskLowPtrBits(const DstOp &Res, const SrcOp &Op0, uint32_t NumBits)
Build and insert Res = G_PTRMASK Op0, G_CONSTANT (1 << NumBits) - 1.
virtual MachineInstrBuilder buildConstant(const DstOp &Res, const ConstantInt &Val)
Build and insert Res = G_CONSTANT Val.
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & addUse(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register use operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
Representation of each machine instruction.
const MachineOperand & getOperand(unsigned i) const
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
Register getReg() const
getReg - Returns the register number.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register createGenericVirtualRegister(LLT Ty, StringRef Name="")
Create and return a new generic virtual register with low-level type Ty.
const TargetRegisterInfo * getTargetRegisterInfo() const
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Register getStackPointerRegisterToSaveRestore() const
If a physical register, this specifies the register that llvm.savestack/llvm.restorestack should save...
Primary interface to the complete machine description for the target machine.
Target - Wrapper for Target specific information.
LLVM Value Representation.
Definition Value.h:75
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ MO_NC
MO_NC - Indicates whether the linker is expected to check the symbol reference for overflow.
@ MO_PAGEOFF
MO_PAGEOFF - A symbol operand with this flag represents the offset of that symbol within a 4K page.
@ MO_GOT
MO_GOT - This flag indicates that a symbol operand represents the address of the GOT entry for the sy...
@ MO_PREL
MO_PREL - Indicates that the bits of the symbol operand represented by MO_G0 etc are PC relative.
@ MO_PAGE
MO_PAGE - A symbol operand with this flag represents the pc-relative offset of the 4K page containing...
@ MO_TAGGED
MO_TAGGED - With MO_PAGE, indicates that the page includes a memory tag in bits 56-63.
@ MO_G3
MO_G3 - A symbol operand with this flag (granule 3) represents the high 16-bits of a 64-bit address,...
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
LLVM_ABI LegalityPredicate scalarOrEltWiderThan(unsigned TypeIdx, unsigned Size)
True iff the specified type index is a scalar or a vector with an element type that's wider than the ...
LLVM_ABI LegalityPredicate isPointerVector(unsigned TypeIdx)
True iff the specified type index is a vector of pointers (with any address space).
LLVM_ABI LegalityPredicate typeInSet(unsigned TypeIdx, std::initializer_list< LLT > TypesInit)
True iff the given type index is one of the specified types.
LLVM_ABI LegalityPredicate smallerThan(unsigned TypeIdx0, unsigned TypeIdx1)
True iff the first type index has a smaller total bit size than second type index.
LLVM_ABI LegalityPredicate atomicOrderingAtLeastOrStrongerThan(unsigned MMOIdx, AtomicOrdering Ordering)
True iff the specified MMO index has at an atomic ordering of at Ordering or stronger.
Predicate any(Predicate P0, Predicate P1)
True iff P0 or P1 are true.
LLVM_ABI LegalityPredicate isVector(unsigned TypeIdx)
True iff the specified type index is a vector.
Predicate all(Predicate P0, Predicate P1)
True iff P0 and P1 are true.
LLVM_ABI LegalityPredicate typeIs(unsigned TypeIdx, LLT TypesInit)
True iff the given type index is the specified type.
LLVM_ABI LegalityPredicate scalarWiderThan(unsigned TypeIdx, unsigned Size)
True iff the specified type index is a scalar that's wider than the given size.
LLVM_ABI LegalityPredicate scalarNarrowerThan(unsigned TypeIdx, unsigned Size)
True iff the specified type index is a scalar that's narrower than the given size.
@ Bitcast
Perform the operation on a different, but equivalently sized type.
LLVM_ABI LegalizeMutation moreElementsToNextPow2(unsigned TypeIdx, unsigned Min=0)
Add more elements to the type for the given type index to the next power of.
LLVM_ABI LegalizeMutation scalarize(unsigned TypeIdx)
Break up the vector type for the given type index into the element type.
LLVM_ABI LegalizeMutation changeElementTo(unsigned TypeIdx, unsigned FromTypeIdx)
Keep the same scalar or element type as the given type index.
LLVM_ABI LegalizeMutation widenScalarOrEltToNextPow2(unsigned TypeIdx, unsigned Min=0)
Widen the scalar type or vector element type for the given type index to the next power of 2.
LLVM_ABI LegalizeMutation changeTo(unsigned TypeIdx, LLT Ty)
Select this specific type for the given type index.
LLVM_ABI LegalizeMutation changeElementSizeTo(unsigned TypeIdx, unsigned FromTypeIdx)
Change the scalar size or element size to have the same scalar size as type index FromIndex.
operand_type_match m_Reg()
ConstantMatch< APInt > m_ICst(APInt &Cst)
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
BinaryOp_match< LHS, RHS, TargetOpcode::G_PTR_ADD, false > m_GPtrAdd(const LHS &L, const RHS &R)
Invariant opcodes: All instruction sets have these as their low opcodes.
This is an optimization pass for GlobalISel generic memory operations.
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
LLVM_ABI std::optional< APInt > isConstantOrConstantSplatVector(Register Def, const MachineRegisterInfo &MRI)
Determines if Def defines a constant integer or a splat vector of constant integers.
Definition Utils.cpp:1517
@ Offset
Definition DWP.cpp:577
LLVM_ABI void constrainSelectedInstRegOperands(MachineInstr &I, const TargetInstrInfo &TII, const TargetRegisterInfo &TRI, const RegisterBankInfo &RBI)
Mutate the newly-selected instruction I to constrain its (possibly generic) virtual register operands...
Definition Utils.cpp:159
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
std::function< bool(const LegalityQuery &)> LegalityPredicate
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
AtomicOrdering
Atomic ordering for LLVM's memory model.
@ Add
Sum of integers.
IntPtrTy
Definition InstrProf.h:82
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
constexpr bool isShiftedInt(int64_t x)
Checks if a signed integer is an N bit number shifted left by S.
Definition MathExtras.h:183
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Align assumeAligned(uint64_t Value)
Treats the value 0 as a 1, so Align is always at least 1.
Definition Alignment.h:100
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
The LegalityQuery object bundles together all the information that's needed to decide whether a given...
ArrayRef< MemDesc > MMODescrs
Operations which require memory can use this to place requirements on the memory type for each MMO.
ArrayRef< LLT > Types
This class contains a discriminated union of information about pointers in memory operands,...