LLVM 24.0.0git
RISCVLegalizerInfo.cpp
Go to the documentation of this file.
1//===-- RISCVLegalizerInfo.cpp ----------------------------------*- C++ -*-===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8/// \file
9/// This file implements the targeting of the Machinelegalizer class for RISC-V.
10/// \todo This should be generated by TableGen.
11//===----------------------------------------------------------------------===//
12
13#include "RISCVLegalizerInfo.h"
16#include "RISCVSubtarget.h"
30#include "llvm/IR/Intrinsics.h"
31#include "llvm/IR/IntrinsicsRISCV.h"
32#include "llvm/IR/Type.h"
33
34using namespace llvm;
35using namespace LegalityPredicates;
36using namespace LegalizeMutations;
37using namespace MIPatternMatch;
38
40typeIsLegalIntOrFPVec(unsigned TypeIdx,
41 std::initializer_list<LLT> IntOrFPVecTys,
42 const RISCVSubtarget &ST) {
43 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
44 return ST.hasVInstructions() &&
45 (Query.Types[TypeIdx].getScalarSizeInBits() != 64 ||
46 ST.hasVInstructionsI64()) &&
47 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
48 ST.getELen() == 64);
49 };
50
51 return all(typeInSet(TypeIdx, IntOrFPVecTys), P);
52}
53
55typeIsLegalBoolVec(unsigned TypeIdx, std::initializer_list<LLT> BoolVecTys,
56 const RISCVSubtarget &ST) {
57 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
58 return ST.hasVInstructions() &&
59 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
60 ST.getELen() == 64);
61 };
62 return all(typeInSet(TypeIdx, BoolVecTys), P);
63}
64
65static LegalityPredicate typeIsLegalPtrVec(unsigned TypeIdx,
66 std::initializer_list<LLT> PtrVecTys,
67 const RISCVSubtarget &ST) {
68 LegalityPredicate P = [=, &ST](const LegalityQuery &Query) {
69 return ST.hasVInstructions() &&
70 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 1 ||
71 ST.getELen() == 64) &&
72 (Query.Types[TypeIdx].getElementCount().getKnownMinValue() != 16 ||
73 Query.Types[TypeIdx].getScalarSizeInBits() == 32);
74 };
75 return all(typeInSet(TypeIdx, PtrVecTys), P);
76}
77
79 : STI(ST), XLen(STI.getXLen()), sXLen(LLT::scalar(XLen)) {
80 const LLT sDoubleXLen = LLT::scalar(2 * XLen);
81 const LLT p0 = LLT::pointer(0, XLen);
82 const LLT s1 = LLT::scalar(1);
83 const LLT s8 = LLT::scalar(8);
84 const LLT s16 = LLT::scalar(16);
85 const LLT f16 = LLT::float16();
86 const LLT s32 = LLT::scalar(32);
87 const LLT s64 = LLT::scalar(64);
88 const LLT s128 = LLT::scalar(128);
89
90 const LLT nxv1s1 = LLT::scalable_vector(1, s1);
91 const LLT nxv2s1 = LLT::scalable_vector(2, s1);
92 const LLT nxv4s1 = LLT::scalable_vector(4, s1);
93 const LLT nxv8s1 = LLT::scalable_vector(8, s1);
94 const LLT nxv16s1 = LLT::scalable_vector(16, s1);
95 const LLT nxv32s1 = LLT::scalable_vector(32, s1);
96 const LLT nxv64s1 = LLT::scalable_vector(64, s1);
97
98 const LLT nxv1s8 = LLT::scalable_vector(1, s8);
99 const LLT nxv2s8 = LLT::scalable_vector(2, s8);
100 const LLT nxv4s8 = LLT::scalable_vector(4, s8);
101 const LLT nxv8s8 = LLT::scalable_vector(8, s8);
102 const LLT nxv16s8 = LLT::scalable_vector(16, s8);
103 const LLT nxv32s8 = LLT::scalable_vector(32, s8);
104 const LLT nxv64s8 = LLT::scalable_vector(64, s8);
105
106 const LLT nxv1s16 = LLT::scalable_vector(1, s16);
107 const LLT nxv2s16 = LLT::scalable_vector(2, s16);
108 const LLT nxv4s16 = LLT::scalable_vector(4, s16);
109 const LLT nxv8s16 = LLT::scalable_vector(8, s16);
110 const LLT nxv16s16 = LLT::scalable_vector(16, s16);
111 const LLT nxv32s16 = LLT::scalable_vector(32, s16);
112
113 const LLT nxv1s32 = LLT::scalable_vector(1, s32);
114 const LLT nxv2s32 = LLT::scalable_vector(2, s32);
115 const LLT nxv4s32 = LLT::scalable_vector(4, s32);
116 const LLT nxv8s32 = LLT::scalable_vector(8, s32);
117 const LLT nxv16s32 = LLT::scalable_vector(16, s32);
118
119 const LLT nxv1s64 = LLT::scalable_vector(1, s64);
120 const LLT nxv2s64 = LLT::scalable_vector(2, s64);
121 const LLT nxv4s64 = LLT::scalable_vector(4, s64);
122 const LLT nxv8s64 = LLT::scalable_vector(8, s64);
123
124 const LLT nxv1p0 = LLT::scalable_vector(1, p0);
125 const LLT nxv2p0 = LLT::scalable_vector(2, p0);
126 const LLT nxv4p0 = LLT::scalable_vector(4, p0);
127 const LLT nxv8p0 = LLT::scalable_vector(8, p0);
128 const LLT nxv16p0 = LLT::scalable_vector(16, p0);
129
130 using namespace TargetOpcode;
131
132 auto BoolVecTys = {nxv1s1, nxv2s1, nxv4s1, nxv8s1, nxv16s1, nxv32s1, nxv64s1};
133
134 auto IntOrFPVecTys = {nxv1s8, nxv2s8, nxv4s8, nxv8s8, nxv16s8, nxv32s8,
135 nxv64s8, nxv1s16, nxv2s16, nxv4s16, nxv8s16, nxv16s16,
136 nxv32s16, nxv1s32, nxv2s32, nxv4s32, nxv8s32, nxv16s32,
137 nxv1s64, nxv2s64, nxv4s64, nxv8s64};
138
139 auto PtrVecTys = {nxv1p0, nxv2p0, nxv4p0, nxv8p0, nxv16p0};
140
141 getActionDefinitionsBuilder({G_ADD, G_SUB})
142 .legalFor({sXLen})
143 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
144 .customFor(ST.is64Bit(), {s32})
146 .clampScalar(0, sXLen, sXLen);
147
148 getActionDefinitionsBuilder({G_AND, G_OR, G_XOR})
149 .legalFor({sXLen})
150 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
152 .clampScalar(0, sXLen, sXLen);
153
155 {G_UADDE, G_UADDO, G_USUBE, G_USUBO, G_READ_REGISTER, G_WRITE_REGISTER})
156 .lower();
157
158 getActionDefinitionsBuilder({G_SADDE, G_SADDO, G_SSUBE, G_SSUBO})
159 .minScalar(0, sXLen)
160 .lower();
161
162 // TODO: Use Vector Single-Width Saturating Instructions for vector types.
164 {G_UADDSAT, G_SADDSAT, G_USUBSAT, G_SSUBSAT, G_SSHLSAT, G_USHLSAT})
165 .lower();
166
167 getActionDefinitionsBuilder({G_SHL, G_ASHR, G_LSHR})
168 .legalFor({{sXLen, sXLen}})
169 .customFor(ST.is64Bit(), {{s32, s32}})
170 .widenScalarToNextPow2(0)
171 .clampScalar(1, sXLen, sXLen)
172 .clampScalar(0, sXLen, sXLen);
173
174 getActionDefinitionsBuilder({G_ZEXT, G_SEXT, G_ANYEXT})
175 .legalFor({{s32, s16}})
176 .legalFor(ST.is64Bit(), {{s64, s16}, {s64, s32}})
177 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
178 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)))
179 .customIf(typeIsLegalBoolVec(1, BoolVecTys, ST))
180 .maxScalar(0, sXLen);
181
182 getActionDefinitionsBuilder(G_TRUNC).alwaysLegal();
183
184 {
185 LegalityPredicate ValidSextInRegWidth = all(sizeIs(0, 64), immIs(0, 32));
186
187 if (STI.hasStdExtZbb())
188 ValidSextInRegWidth =
189 LegalityPredicates::any(ValidSextInRegWidth, immInSet(0, {8, 16}));
190
191 getActionDefinitionsBuilder(G_SEXT_INREG)
192 .legalIf(all(typeIs(0, sXLen), ValidSextInRegWidth))
193 .clampScalar(0, sXLen, sXLen)
194 .lower();
195 }
196
197 // Merge/Unmerge
198 for (unsigned Op : {G_MERGE_VALUES, G_UNMERGE_VALUES}) {
199 auto &MergeUnmergeActions = getActionDefinitionsBuilder(Op);
200 unsigned BigTyIdx = Op == G_MERGE_VALUES ? 0 : 1;
201 unsigned LitTyIdx = Op == G_MERGE_VALUES ? 1 : 0;
202 if (XLen == 32 && ST.hasStdExtD()) {
203 MergeUnmergeActions.legalIf(
204 all(typeIs(BigTyIdx, s64), typeIs(LitTyIdx, s32)));
205 }
206 MergeUnmergeActions.widenScalarToNextPow2(LitTyIdx, XLen)
207 .widenScalarToNextPow2(BigTyIdx, XLen)
208 .clampScalar(LitTyIdx, sXLen, sXLen)
209 .clampScalar(BigTyIdx, sXLen, sXLen);
210 }
211
212 getActionDefinitionsBuilder({G_FSHL, G_FSHR}).lower();
213
214 getActionDefinitionsBuilder({G_ROTR, G_ROTL})
215 .legalFor(ST.hasStdExtZbb() || ST.hasStdExtZbkb(), {{sXLen, sXLen}})
216 .customFor(ST.is64Bit() && (ST.hasStdExtZbb() || ST.hasStdExtZbkb()),
217 {{s32, s32}})
218 .lower();
219
220 getActionDefinitionsBuilder(G_BITREVERSE)
221 .customFor(ST.hasStdExtZbkb(), {s8})
222 .maxScalar(0, sXLen)
223 .lower();
224
225 getActionDefinitionsBuilder(G_BITCAST).legalIf(
227 typeIsLegalBoolVec(0, BoolVecTys, ST)),
229 typeIsLegalBoolVec(1, BoolVecTys, ST))));
230
231 auto &BSWAPActions = getActionDefinitionsBuilder(G_BSWAP);
232 if (ST.hasStdExtZbb() || ST.hasStdExtZbkb())
233 BSWAPActions.legalFor({sXLen}).clampScalar(0, sXLen, sXLen);
234 else
235 BSWAPActions.maxScalar(0, sXLen).lower();
236
237 getActionDefinitionsBuilder(G_CLMUL)
238 .legalFor(ST.hasStdExtZbkc(), {sXLen})
239 .unsupported();
240
241 getActionDefinitionsBuilder(G_CLMULH)
242 .legalFor(ST.hasStdExtZbkc(), {sXLen})
243 .customFor(ST.is64Bit() && ST.hasStdExtZbkc(), {s32})
244 .unsupported();
245
246 // CLMULR is Zbc-only; Zbkc is a subset that has CLMUL/CLMULH but not CLMULR.
247 getActionDefinitionsBuilder(G_CLMULR)
248 .legalFor(ST.hasStdExtZbc(), {sXLen})
249 .customFor(ST.is64Bit() && ST.hasStdExtZbc(), {s32})
250 .unsupported();
251
252 auto &CountZerosActions = getActionDefinitionsBuilder({G_CTLZ, G_CTTZ});
253 auto &CountZerosPoisonActions =
254 getActionDefinitionsBuilder({G_CTLZ_ZERO_POISON, G_CTTZ_ZERO_POISON});
255 if (ST.hasStdExtZbb()) {
256 CountZerosActions.legalFor({{sXLen, sXLen}})
257 .customFor({{s32, s32}})
258 .clampScalar(0, s32, sXLen)
259 .widenScalarToNextPow2(0)
260 .scalarSameSizeAs(1, 0);
261 } else {
262 CountZerosActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0).lower();
263 CountZerosPoisonActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0);
264 }
265 CountZerosPoisonActions.lower();
266
267 auto &CountSignActions = getActionDefinitionsBuilder(G_CTLS);
268 if (ST.hasStdExtP()) {
269 CountSignActions.legalFor({{sXLen, sXLen}})
270 .customFor({{s32, s32}})
271 .clampScalar(0, s32, sXLen)
272 .widenScalarToNextPow2(0)
273 .scalarSameSizeAs(1, 0);
274 } else {
275 CountSignActions.maxScalar(0, sXLen).scalarSameSizeAs(1, 0).lower();
276 }
277
278 auto &CTPOPActions = getActionDefinitionsBuilder(G_CTPOP);
279 if (ST.hasStdExtZbb()) {
280 CTPOPActions.legalFor({{sXLen, sXLen}})
281 .clampScalar(0, sXLen, sXLen)
282 .scalarSameSizeAs(1, 0);
283 } else {
284 CTPOPActions.widenScalarToNextPow2(0, /*Min*/ 8)
285 .clampScalar(0, s8, sXLen)
286 .scalarSameSizeAs(1, 0)
287 .lower();
288 }
289
290 getActionDefinitionsBuilder(G_CONSTANT)
291 .legalFor({p0})
292 .legalFor(!ST.is64Bit(), {s32})
293 .customFor(ST.is64Bit(), {s64})
294 .widenScalarToNextPow2(0)
295 .clampScalar(0, sXLen, sXLen);
296
297 // TODO: transform illegal vector types into legal vector type
298 getActionDefinitionsBuilder(G_FREEZE)
299 .legalFor({s16, s32, p0})
300 .legalFor(ST.is64Bit(), {s64})
301 .legalIf(typeIsLegalBoolVec(0, BoolVecTys, ST))
302 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
303 .widenScalarToNextPow2(0)
304 .clampScalar(0, s16, sXLen);
305
306 // TODO: transform illegal vector types into legal vector type
307 // TODO: Merge with G_FREEZE?
308 getActionDefinitionsBuilder(
309 {G_IMPLICIT_DEF, G_CONSTANT_FOLD_BARRIER})
310 .legalFor({s32, sXLen, p0})
311 .legalIf(typeIsLegalBoolVec(0, BoolVecTys, ST))
312 .legalIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST))
313 .widenScalarToNextPow2(0)
314 .clampScalar(0, s32, sXLen);
315
316 getActionDefinitionsBuilder(G_ICMP)
317 .legalFor({{sXLen, sXLen}, {sXLen, p0}})
318 .legalIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST),
319 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)))
320 .widenScalarOrEltToNextPow2OrMinSize(1, 8)
321 .clampScalar(1, sXLen, sXLen)
322 .clampScalar(0, sXLen, sXLen);
323
324 getActionDefinitionsBuilder(G_SELECT)
325 .legalFor({{s32, sXLen}, {p0, sXLen}})
326 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
327 typeIsLegalBoolVec(1, BoolVecTys, ST)))
328 .legalFor(XLen == 64 || ST.hasStdExtD(), {{s64, sXLen}})
329 .widenScalarToNextPow2(0)
330 .clampScalar(0, s32, (XLen == 64 || ST.hasStdExtD()) ? s64 : s32)
331 .clampScalar(1, sXLen, sXLen);
332
333 auto &LoadActions = getActionDefinitionsBuilder(G_LOAD);
334 auto &StoreActions = getActionDefinitionsBuilder(G_STORE);
335 auto &ExtLoadActions = getActionDefinitionsBuilder({G_SEXTLOAD, G_ZEXTLOAD});
336
337 // Return the alignment needed for scalar memory ops. If unaligned scalar mem
338 // is supported, we only require byte alignment. Otherwise, we need the memory
339 // op to be natively aligned.
340 auto getScalarMemAlign = [&ST](unsigned Size) {
341 return ST.enableUnalignedScalarMem() ? 8 : Size;
342 };
343
344 LoadActions.legalForTypesWithMemDesc(
345 {{s16, p0, s8, getScalarMemAlign(8)},
346 {s32, p0, s8, getScalarMemAlign(8)},
347 {s16, p0, s16, getScalarMemAlign(16)},
348 {s32, p0, s16, getScalarMemAlign(16)},
349 {s32, p0, s32, getScalarMemAlign(32)},
350 {p0, p0, sXLen, getScalarMemAlign(XLen)}});
351 StoreActions.legalForTypesWithMemDesc(
352 {{s16, p0, s8, getScalarMemAlign(8)},
353 {s32, p0, s8, getScalarMemAlign(8)},
354 {s16, p0, s16, getScalarMemAlign(16)},
355 {s32, p0, s16, getScalarMemAlign(16)},
356 {s32, p0, s32, getScalarMemAlign(32)},
357 {p0, p0, sXLen, getScalarMemAlign(XLen)}});
358 ExtLoadActions.legalForTypesWithMemDesc(
359 {{sXLen, p0, s8, getScalarMemAlign(8)},
360 {sXLen, p0, s16, getScalarMemAlign(16)}});
361 if (XLen == 64) {
362 LoadActions.legalForTypesWithMemDesc(
363 {{s64, p0, s8, getScalarMemAlign(8)},
364 {s64, p0, s16, getScalarMemAlign(16)},
365 {s64, p0, s32, getScalarMemAlign(32)},
366 {s64, p0, s64, getScalarMemAlign(64)}});
367 StoreActions.legalForTypesWithMemDesc(
368 {{s64, p0, s8, getScalarMemAlign(8)},
369 {s64, p0, s16, getScalarMemAlign(16)},
370 {s64, p0, s32, getScalarMemAlign(32)},
371 {s64, p0, s64, getScalarMemAlign(64)}});
372 ExtLoadActions.legalForTypesWithMemDesc(
373 {{s64, p0, s32, getScalarMemAlign(32)}});
374 } else if (ST.hasStdExtD()) {
375 LoadActions.legalForTypesWithMemDesc(
376 {{s64, p0, s64, getScalarMemAlign(64)}});
377 StoreActions.legalForTypesWithMemDesc(
378 {{s64, p0, s64, getScalarMemAlign(64)}});
379 }
380
381 // Vector loads/stores.
382 if (ST.hasVInstructions()) {
383 LoadActions.legalForTypesWithMemDesc({{nxv2s8, p0, nxv2s8, 8},
384 {nxv4s8, p0, nxv4s8, 8},
385 {nxv8s8, p0, nxv8s8, 8},
386 {nxv16s8, p0, nxv16s8, 8},
387 {nxv32s8, p0, nxv32s8, 8},
388 {nxv64s8, p0, nxv64s8, 8},
389 {nxv2s16, p0, nxv2s16, 16},
390 {nxv4s16, p0, nxv4s16, 16},
391 {nxv8s16, p0, nxv8s16, 16},
392 {nxv16s16, p0, nxv16s16, 16},
393 {nxv32s16, p0, nxv32s16, 16},
394 {nxv2s32, p0, nxv2s32, 32},
395 {nxv4s32, p0, nxv4s32, 32},
396 {nxv8s32, p0, nxv8s32, 32},
397 {nxv16s32, p0, nxv16s32, 32}});
398 StoreActions.legalForTypesWithMemDesc({{nxv2s8, p0, nxv2s8, 8},
399 {nxv4s8, p0, nxv4s8, 8},
400 {nxv8s8, p0, nxv8s8, 8},
401 {nxv16s8, p0, nxv16s8, 8},
402 {nxv32s8, p0, nxv32s8, 8},
403 {nxv64s8, p0, nxv64s8, 8},
404 {nxv2s16, p0, nxv2s16, 16},
405 {nxv4s16, p0, nxv4s16, 16},
406 {nxv8s16, p0, nxv8s16, 16},
407 {nxv16s16, p0, nxv16s16, 16},
408 {nxv32s16, p0, nxv32s16, 16},
409 {nxv2s32, p0, nxv2s32, 32},
410 {nxv4s32, p0, nxv4s32, 32},
411 {nxv8s32, p0, nxv8s32, 32},
412 {nxv16s32, p0, nxv16s32, 32}});
413
414 if (ST.getELen() == 64) {
415 LoadActions.legalForTypesWithMemDesc({{nxv1s8, p0, nxv1s8, 8},
416 {nxv1s16, p0, nxv1s16, 16},
417 {nxv1s32, p0, nxv1s32, 32}});
418 StoreActions.legalForTypesWithMemDesc({{nxv1s8, p0, nxv1s8, 8},
419 {nxv1s16, p0, nxv1s16, 16},
420 {nxv1s32, p0, nxv1s32, 32}});
421 }
422
423 if (ST.hasVInstructionsI64()) {
424 LoadActions.legalForTypesWithMemDesc({{nxv1s64, p0, nxv1s64, 64},
425 {nxv2s64, p0, nxv2s64, 64},
426 {nxv4s64, p0, nxv4s64, 64},
427 {nxv8s64, p0, nxv8s64, 64}});
428 StoreActions.legalForTypesWithMemDesc({{nxv1s64, p0, nxv1s64, 64},
429 {nxv2s64, p0, nxv2s64, 64},
430 {nxv4s64, p0, nxv4s64, 64},
431 {nxv8s64, p0, nxv8s64, 64}});
432 }
433
434 // we will take the custom lowering logic if we have scalable vector types
435 // with non-standard alignments
436 LoadActions.customIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST));
437 StoreActions.customIf(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST));
438
439 // Pointers require that XLen sized elements are legal.
440 if (XLen <= ST.getELen()) {
441 LoadActions.customIf(typeIsLegalPtrVec(0, PtrVecTys, ST));
442 StoreActions.customIf(typeIsLegalPtrVec(0, PtrVecTys, ST));
443 }
444 }
445
446 LoadActions.widenScalarToNextPow2(0, /* MinSize = */ 8)
447 .lowerIfMemSizeNotByteSizePow2()
448 .clampScalar(0, s16, sXLen)
449 .lower();
450 StoreActions
451 .clampScalar(0, s16, sXLen)
452 .lowerIfMemSizeNotByteSizePow2()
453 .lower();
454
455 ExtLoadActions.widenScalarToNextPow2(0).clampScalar(0, sXLen, sXLen).lower();
456
457 getActionDefinitionsBuilder({G_PTR_ADD, G_PTRMASK}).legalFor({{p0, sXLen}});
458
459 getActionDefinitionsBuilder(G_PTRTOINT)
460 .legalFor({{sXLen, p0}})
461 .clampScalar(0, sXLen, sXLen);
462
463 getActionDefinitionsBuilder(G_INTTOPTR)
464 .legalFor({{p0, sXLen}})
465 .clampScalar(1, sXLen, sXLen);
466
467 getActionDefinitionsBuilder(G_BR).alwaysLegal();
468
469 getActionDefinitionsBuilder(G_BRCOND).legalFor({sXLen}).minScalar(0, sXLen);
470
471 getActionDefinitionsBuilder(G_BRJT).customFor({{p0, sXLen}});
472
473 getActionDefinitionsBuilder(G_BRINDIRECT).legalFor({p0});
474
475 getActionDefinitionsBuilder(G_PHI)
476 .legalFor({p0, s32, sXLen})
477 .widenScalarToNextPow2(0)
478 .clampScalar(0, s32, sXLen);
479
480 getActionDefinitionsBuilder({G_GLOBAL_VALUE, G_JUMP_TABLE, G_CONSTANT_POOL})
481 .legalFor({p0});
482
483 if (ST.hasStdExtZmmul()) {
484 getActionDefinitionsBuilder(G_MUL)
485 .legalFor({sXLen})
486 .widenScalarToNextPow2(0)
487 .clampScalar(0, sXLen, sXLen);
488
489 // clang-format off
490 getActionDefinitionsBuilder({G_SMULH, G_UMULH})
491 .legalFor({sXLen})
492 .lower();
493 // clang-format on
494
495 getActionDefinitionsBuilder({G_SMULO, G_UMULO}).minScalar(0, sXLen).lower();
496 } else {
497 getActionDefinitionsBuilder(G_MUL)
498 .libcallFor({sXLen, sDoubleXLen})
499 .widenScalarToNextPow2(0)
500 .clampScalar(0, sXLen, sDoubleXLen);
501
502 getActionDefinitionsBuilder({G_SMULH, G_UMULH}).lowerFor({sXLen});
503
504 getActionDefinitionsBuilder({G_SMULO, G_UMULO})
505 .minScalar(0, sXLen)
506 // Widen sXLen to sDoubleXLen so we can use a single libcall to get
507 // the low bits for the mul result and high bits to do the overflow
508 // check.
509 .widenScalarIf(typeIs(0, sXLen),
510 LegalizeMutations::changeTo(0, sDoubleXLen))
511 .lower();
512 }
513
514 if (ST.hasStdExtM()) {
515 getActionDefinitionsBuilder({G_SDIV, G_UDIV, G_UREM})
516 .legalFor({sXLen})
517 .customFor({s32})
518 .libcallFor({sDoubleXLen})
519 .clampScalar(0, s32, sDoubleXLen)
520 .widenScalarToNextPow2(0);
521 getActionDefinitionsBuilder(G_SREM)
522 .legalFor({sXLen})
523 .libcallFor({sDoubleXLen})
524 .clampScalar(0, sXLen, sDoubleXLen)
525 .widenScalarToNextPow2(0);
526 } else {
527 getActionDefinitionsBuilder({G_UDIV, G_SDIV, G_UREM, G_SREM})
528 .libcallFor({sXLen, sDoubleXLen})
529 .clampScalar(0, sXLen, sDoubleXLen)
530 .widenScalarToNextPow2(0);
531 }
532
533 // TODO: Use libcall for sDoubleXLen.
534 getActionDefinitionsBuilder({G_SDIVREM, G_UDIVREM}).lower();
535
536 getActionDefinitionsBuilder(G_ABS)
537 .customFor(ST.hasStdExtZbb(), {sXLen})
538 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
539 .lower();
540
541 getActionDefinitionsBuilder({G_ABDS, G_ABDU})
542 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
543 .lower();
544
545 getActionDefinitionsBuilder({G_UMAX, G_UMIN, G_SMAX, G_SMIN})
546 .legalFor(ST.hasStdExtZbb(), {sXLen})
547 .minScalar(ST.hasStdExtZbb(), 0, sXLen)
548 .lower();
549
550 getActionDefinitionsBuilder({G_SCMP, G_UCMP}).lower();
551
552 getActionDefinitionsBuilder(G_FRAME_INDEX).legalFor({p0});
553
554 getActionDefinitionsBuilder({G_MEMCPY, G_MEMMOVE, G_MEMSET}).libcall();
555
556 getActionDefinitionsBuilder({G_MEMCPY_INLINE, G_MEMSET_INLINE}).lower();
557
558 getActionDefinitionsBuilder({G_DYN_STACKALLOC, G_STACKSAVE, G_STACKRESTORE})
559 .lower();
560
561 // On RV64 the 64-bit counter CSRs (cycle/time) are read directly. On RV32
562 // they are custom-legally lowered to a re-read-the-high-half loop (see
563 // legalizeReadCounter).
564 getActionDefinitionsBuilder({G_READCYCLECOUNTER, G_READSTEADYCOUNTER})
565 .legalFor(ST.is64Bit(), {s64})
566 .customFor(!ST.is64Bit(), {s64});
567
568 // FP Operations
569
570 // FIXME: Support s128 for rv32 when libcall handling is able to use sret.
571 getActionDefinitionsBuilder({G_FADD, G_FSUB, G_FMUL, G_FDIV, G_FMA, G_FSQRT,
572 G_FMAXNUM, G_FMINNUM, G_FMAXIMUMNUM,
573 G_FMINIMUMNUM})
574 .legalFor(ST.hasStdExtF(), {s32})
575 .legalFor(ST.hasStdExtD(), {s64})
576 .legalFor(ST.hasStdExtZfh(), {s16})
577 .libcallFor({s32, s64})
578 .libcallFor(ST.is64Bit(), {s128});
579
580 getActionDefinitionsBuilder({G_FNEG, G_FABS})
581 .legalFor(ST.hasStdExtF(), {s32})
582 .legalFor(ST.hasStdExtD(), {s64})
583 .legalFor(ST.hasStdExtZfh(), {s16})
584 .lowerFor({s32, s64, s128});
585
586 getActionDefinitionsBuilder(G_FREM)
587 .libcallFor({s32, s64})
588 .libcallFor(ST.is64Bit(), {s128})
589 .minScalar(0, s32)
590 .scalarize(0);
591
592 getActionDefinitionsBuilder(G_FCOPYSIGN)
593 .legalFor(ST.hasStdExtF(), {{s32, s32}})
594 .legalFor(ST.hasStdExtD(), {{s64, s64}, {s32, s64}, {s64, s32}})
595 .legalFor(ST.hasStdExtZfh(), {{s16, s16}, {s16, s32}, {s32, s16}})
596 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s16, s64}, {s64, s16}})
597 .lower();
598
599 // FIXME: Use Zfhmin.
600 getActionDefinitionsBuilder(G_FPTRUNC)
601 .legalFor(ST.hasStdExtD(), {{s32, s64}})
602 .legalFor(ST.hasStdExtZfh(), {{s16, s32}})
603 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s16, s64}})
604 .libcallFor({{s32, s64}})
605 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}});
606 getActionDefinitionsBuilder(G_FPEXT)
607 .legalFor(ST.hasStdExtD(), {{s64, s32}})
608 .legalFor(ST.hasStdExtZfhmin(), {{s32, s16}})
609 .legalFor(ST.hasStdExtZfh() && ST.hasStdExtD(), {{s64, s16}})
610 .libcallFor(!ST.hasStdExtZfhmin(), {{s32, s16}})
611 .libcallFor({{s64, s32}})
612 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}});
613
614 getActionDefinitionsBuilder(G_FCMP)
615 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
616 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
617 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
618 .clampScalar(0, sXLen, sXLen)
619 .libcallFor({{sXLen, s32}, {sXLen, s64}})
620 .libcallFor(ST.is64Bit(), {{sXLen, s128}});
621
622 // TODO: Support vector version of G_IS_FPCLASS.
623 getActionDefinitionsBuilder(G_IS_FPCLASS)
624 .customFor(ST.hasStdExtF(), {{s1, s32}})
625 .customFor(ST.hasStdExtD(), {{s1, s64}})
626 .customFor(ST.hasStdExtZfh(), {{s1, s16}})
627 .lower();
628
629 getActionDefinitionsBuilder(G_FCONSTANT)
630 .legalFor(ST.hasStdExtF(), {s32})
631 .legalFor(ST.hasStdExtD(), {s64})
632 .legalFor(ST.hasStdExtZfh(), {s16})
633 .customFor(!ST.is64Bit(), {s32})
634 .customFor(ST.is64Bit(), {s32, s64})
635 .lowerFor({s64, s128});
636
637 getActionDefinitionsBuilder({G_FPTOSI, G_FPTOUI})
638 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
639 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
640 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
641 .customFor(ST.is64Bit() && ST.hasStdExtF(), {{s32, s32}})
642 .customFor(ST.is64Bit() && ST.hasStdExtD(), {{s32, s64}})
643 .customFor(ST.is64Bit() && ST.hasStdExtZfh(), {{s32, s16}})
644 .widenScalarToNextPow2(0)
645 .minScalar(0, s32)
646 // The magnitude of a half is at most 65504, so with Zfh use fcvt.w[u].h
647 // and extend the i32 result. Otherwise promote the half source to float
648 // (via fcvt.s.h with Zfhmin, __extendhfsf2 without) and use the float
649 // conversion. On RV32, exclude i64 results here so that they get
650 // narrowed to i32 first.
651 .widenScalarIf(
652 [=, &ST](const LegalityQuery &Query) {
653 return Query.Types[1] == f16 && !ST.hasStdExtZfh() &&
654 (ST.is64Bit() || Query.Types[0] == s32);
655 },
656 changeTo(1, s32))
657 .libcallFor(!ST.hasStdExtZfhmin(), {{s64, f16}})
658 .narrowScalarFor({{s64, f16}}, changeTo(0, s32))
659 .libcallFor({{s32, s32}, {s64, s32}, {s32, s64}, {s64, s64}})
660 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}}) // FIXME RV32.
661 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}, {s128, s128}});
662
663 getActionDefinitionsBuilder({G_LROUND, G_LLROUND})
664 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
665 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
666 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
667 .customFor(ST.is64Bit() && ST.hasStdExtF(), {{s32, s32}})
668 .customFor(ST.is64Bit() && ST.hasStdExtD(), {{s32, s64}})
669 .customFor(ST.is64Bit() && ST.hasStdExtZfh(), {{s32, s16}})
670 .widenScalarIf(typeIs(1, s16), LegalizeMutations::changeTo(1, s32))
671 .libcallFor({{s32, s32},
672 {s64, s32},
673 {s32, s64},
674 {s64, s64},
675 {s32, s128},
676 {s64, s128}});
677
678 getActionDefinitionsBuilder({G_INTRINSIC_LRINT, G_INTRINSIC_LLRINT})
679 .legalFor(ST.hasStdExtF(), {{sXLen, s32}})
680 .legalFor(ST.hasStdExtD(), {{sXLen, s64}})
681 .legalFor(ST.hasStdExtZfh(), {{sXLen, s16}})
682 .minScalar(0, sXLen)
683 .widenScalarIf(typeIs(1, s16), LegalizeMutations::changeTo(1, s32))
684 .libcallFor({{s32, s32},
685 {s64, s32},
686 {s32, s64},
687 {s64, s64},
688 {s32, s128},
689 {s64, s128}});
690
691 getActionDefinitionsBuilder({G_SITOFP, G_UITOFP})
692 .legalFor(ST.hasStdExtF(), {{s32, sXLen}})
693 .legalFor(ST.hasStdExtD(), {{s64, sXLen}})
694 .legalFor(ST.hasStdExtZfh(), {{s16, sXLen}})
695 .widenScalarToNextPow2(1)
696 // Promote to XLen if the operation is legal.
697 .widenScalarIf(
698 [=, &ST](const LegalityQuery &Query) {
699 return Query.Types[0].isScalar() && Query.Types[1].isScalar() &&
700 (Query.Types[1].getSizeInBits() < ST.getXLen()) &&
701 ((ST.hasStdExtF() && Query.Types[0].getSizeInBits() == 32) ||
702 (ST.hasStdExtD() && Query.Types[0].getSizeInBits() == 64) ||
703 (ST.hasStdExtZfh() &&
704 Query.Types[0].getSizeInBits() == 16));
705 },
707 // Otherwise only promote to s32 since we have si libcalls.
708 .minScalar(1, s32)
709 .libcallFor({{s32, s32}, {s64, s32}, {s32, s64}, {s64, s64}})
710 .libcallFor(ST.is64Bit(), {{s128, s32}, {s128, s64}}) // FIXME RV32.
711 .libcallFor(ST.is64Bit(), {{s32, s128}, {s64, s128}, {s128, s128}});
712
713 // FIXME: We can do custom inline expansion like SelectionDAG.
714 getActionDefinitionsBuilder({G_FCEIL, G_FFLOOR, G_FRINT, G_FNEARBYINT,
715 G_INTRINSIC_TRUNC, G_INTRINSIC_ROUND,
716 G_INTRINSIC_ROUNDEVEN})
717 .legalFor(ST.hasStdExtZfa(), {s32})
718 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtD(), {s64})
719 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtZfh(), {s16})
720 .libcallFor({s32, s64})
721 .libcallFor(ST.is64Bit(), {s128});
722
723 getActionDefinitionsBuilder({G_FMAXIMUM, G_FMINIMUM})
724 .legalFor(ST.hasStdExtZfa(), {s32})
725 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtD(), {s64})
726 .legalFor(ST.hasStdExtZfa() && ST.hasStdExtZfh(), {s16});
727
728 getActionDefinitionsBuilder({G_FCOS, G_FSIN, G_FTAN, G_FPOW, G_FLOG, G_FLOG2,
729 G_FLOG10, G_FEXP, G_FEXP2, G_FEXP10, G_FACOS,
730 G_FASIN, G_FATAN, G_FATAN2, G_FCOSH, G_FSINH,
731 G_FTANH, G_FMODF})
732 .libcallFor({s32, s64})
733 .libcallFor(ST.is64Bit(), {s128});
734 getActionDefinitionsBuilder({G_FPOWI, G_FLDEXP})
735 .libcallFor({{s32, s32}, {s64, s32}})
736 .libcallFor(ST.is64Bit(), {s128, s32});
737
738 getActionDefinitionsBuilder(G_FCANONICALIZE)
739 .legalFor(ST.hasStdExtF(), {s32})
740 .legalFor(ST.hasStdExtD(), {s64})
741 .legalFor(ST.hasStdExtZfh(), {s16});
742
743 getActionDefinitionsBuilder(G_VASTART).customFor({p0});
744
745 // va_list must be a pointer, but most sized types are pretty easy to handle
746 // as the destination.
747 getActionDefinitionsBuilder(G_VAARG)
748 // TODO: Implement narrowScalar and widenScalar for G_VAARG for types
749 // other than sXLen.
750 .clampScalar(0, sXLen, sXLen)
751 .lowerForCartesianProduct({sXLen, p0}, {p0});
752
753 getActionDefinitionsBuilder(G_VSCALE)
754 .clampScalar(0, sXLen, sXLen)
755 .customFor({sXLen});
756
757 auto &SplatActions =
758 getActionDefinitionsBuilder(G_SPLAT_VECTOR)
759 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
760 typeIs(1, sXLen)))
761 .customIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST), typeIs(1, s1)));
762 // Handle case of s64 element vectors on RV32. If the subtarget does not have
763 // f64, then try to lower it to G_SPLAT_VECTOR_SPLIT_64_VL. If the subtarget
764 // does have f64, then we don't know whether the type is an f64 or an i64,
765 // so mark the G_SPLAT_VECTOR as legal and decide later what to do with it,
766 // depending on how the instructions it consumes are legalized. They are not
767 // legalized yet since legalization is in reverse postorder, so we cannot
768 // make the decision at this moment.
769 if (XLen == 32) {
770 if (ST.hasVInstructionsF64() && ST.hasStdExtD())
771 SplatActions.legalIf(all(
772 typeInSet(0, {nxv1s64, nxv2s64, nxv4s64, nxv8s64}), typeIs(1, s64)));
773 else if (ST.hasVInstructionsI64())
774 SplatActions.customIf(all(
775 typeInSet(0, {nxv1s64, nxv2s64, nxv4s64, nxv8s64}), typeIs(1, s64)));
776 }
777
778 SplatActions.clampScalar(1, sXLen, sXLen);
779
780 LegalityPredicate ExtractSubvecBitcastPred = [=](const LegalityQuery &Query) {
781 LLT DstTy = Query.Types[0];
782 LLT SrcTy = Query.Types[1];
783 return DstTy.getElementType() == LLT::scalar(1) &&
784 DstTy.getElementCount().getKnownMinValue() >= 8 &&
785 SrcTy.getElementCount().getKnownMinValue() >= 8;
786 };
787 getActionDefinitionsBuilder(G_EXTRACT_SUBVECTOR)
788 // We don't have the ability to slide mask vectors down indexed by their
789 // i1 elements; the smallest we can do is i8. Often we are able to bitcast
790 // to equivalent i8 vectors.
791 .bitcastIf(
792 all(typeIsLegalBoolVec(0, BoolVecTys, ST),
793 typeIsLegalBoolVec(1, BoolVecTys, ST), ExtractSubvecBitcastPred),
794 [=](const LegalityQuery &Query) {
795 LLT CastTy = LLT::vector(
796 Query.Types[0].getElementCount().divideCoefficientBy(8), 8);
797 return std::pair(0, CastTy);
798 })
799 .customIf(LegalityPredicates::any(
800 all(typeIsLegalBoolVec(0, BoolVecTys, ST),
801 typeIsLegalBoolVec(1, BoolVecTys, ST)),
802 all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
803 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST))));
804
805 getActionDefinitionsBuilder(G_INSERT_SUBVECTOR)
806 .customIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST),
807 typeIsLegalBoolVec(1, BoolVecTys, ST)))
808 .customIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
809 typeIsLegalIntOrFPVec(1, IntOrFPVecTys, ST)));
810
811 getActionDefinitionsBuilder(G_ATOMIC_CMPXCHG_WITH_SUCCESS)
812 .lowerIf(all(typeInSet(0, {s8, s16, s32, s64}), typeIs(2, p0)));
813
814 getActionDefinitionsBuilder({G_ATOMIC_CMPXCHG, G_ATOMICRMW_ADD,
815 G_ATOMICRMW_XCHG, G_ATOMICRMW_AND,
816 G_ATOMICRMW_OR, G_ATOMICRMW_XOR})
817 .legalFor(ST.hasStdExtA(), {{sXLen, p0}})
818 .libcallFor(!ST.hasStdExtA(), {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}})
819 .clampScalar(0, sXLen, sXLen);
820
821 getActionDefinitionsBuilder(G_ATOMICRMW_SUB)
822 .libcallFor(!ST.hasStdExtA(), {{s8, p0}, {s16, p0}, {s32, p0}, {s64, p0}})
823 .clampScalar(0, sXLen, sXLen)
824 .lower();
825
826 getActionDefinitionsBuilder(
827 {G_ATOMICRMW_MAX, G_ATOMICRMW_MIN, G_ATOMICRMW_UMAX, G_ATOMICRMW_UMIN})
828 .legalFor(ST.hasStdExtA(), {{sXLen, p0}})
829 .clampScalar(0, sXLen, sXLen)
830 .unsupported();
831
832 getActionDefinitionsBuilder(G_PREFETCH).legalIf(typeIs(0, p0));
833
834 LegalityPredicate InsertVectorEltPred = [=](const LegalityQuery &Query) {
835 LLT VecTy = Query.Types[0];
836 LLT EltTy = Query.Types[1];
837 return VecTy.getElementType() == EltTy;
838 };
839
840 getActionDefinitionsBuilder(G_INSERT_VECTOR_ELT)
841 .legalIf(all(typeIsLegalIntOrFPVec(0, IntOrFPVecTys, ST),
842 InsertVectorEltPred, typeIs(2, sXLen)))
843 .legalIf(all(typeIsLegalBoolVec(0, BoolVecTys, ST), InsertVectorEltPred,
844 typeIs(2, sXLen)));
845
846 getActionDefinitionsBuilder({G_INTRINSIC, G_INTRINSIC_W_SIDE_EFFECTS})
847 .alwaysLegal();
848
849 getActionDefinitionsBuilder(G_FENCE).alwaysLegal();
850
851 getActionDefinitionsBuilder({G_TRAP, G_DEBUGTRAP, G_UBSANTRAP}).alwaysLegal();
852
853 verify(*ST.getInstrInfo());
854}
855
857 MachineInstr &MI) const {
858 Intrinsic::ID IntrinsicID = cast<GIntrinsic>(MI).getIntrinsicID();
859
861 RISCVVIntrinsicsTable::getRISCVVIntrinsicInfo(IntrinsicID)) {
862 if (II->hasScalarOperand() && !II->IsFPIntrinsic) {
863 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
864 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
865
866 auto OldScalar = MI.getOperand(II->ScalarOperand + 2).getReg();
867 // Legalize integer vx form intrinsic.
868 if (MRI.getType(OldScalar).isScalar()) {
869 if (MRI.getType(OldScalar).getSizeInBits() < sXLen.getSizeInBits()) {
870 Helper.Observer.changingInstr(MI);
871 Helper.widenScalarSrc(MI, sXLen, II->ScalarOperand + 2,
872 TargetOpcode::G_ANYEXT);
873 Helper.Observer.changedInstr(MI);
874 } else if (MRI.getType(OldScalar).getSizeInBits() >
875 sXLen.getSizeInBits()) {
876 // TODO: i64 in riscv32.
877 return false;
878 }
879 }
880 }
881 return true;
882 }
883
884 switch (IntrinsicID) {
885 default:
886 return false;
887 case Intrinsic::riscv_clmulh:
888 Helper.MIRBuilder.buildInstr(TargetOpcode::G_CLMULH, {MI.getOperand(0)},
889 {MI.getOperand(2), MI.getOperand(3)});
890 MI.eraseFromParent();
891 return true;
892 case Intrinsic::riscv_clmulr:
893 Helper.MIRBuilder.buildInstr(TargetOpcode::G_CLMULR, {MI.getOperand(0)},
894 {MI.getOperand(2), MI.getOperand(3)});
895 MI.eraseFromParent();
896 return true;
897 case Intrinsic::vacopy: {
898 // vacopy arguments must be legal because of the intrinsic signature.
899 // No need to check here.
900
901 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
902 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
903 MachineFunction &MF = *MI.getMF();
904 const DataLayout &DL = MIRBuilder.getDataLayout();
905 LLVMContext &Ctx = MF.getFunction().getContext();
906
907 Register DstLst = MI.getOperand(1).getReg();
908 LLT PtrTy = MRI.getType(DstLst);
909
910 // Load the source va_list
911 Align Alignment = DL.getABITypeAlign(getTypeForLLT(PtrTy, Ctx));
913 MachinePointerInfo(), MachineMemOperand::MOLoad, PtrTy, Alignment);
914 auto Tmp = MIRBuilder.buildLoad(PtrTy, MI.getOperand(2), *LoadMMO);
915
916 // Store the result in the destination va_list
919 MIRBuilder.buildStore(Tmp, DstLst, *StoreMMO);
920
921 MI.eraseFromParent();
922 return true;
923 }
924 case Intrinsic::riscv_vsetvli:
925 case Intrinsic::riscv_vsetvlimax:
926 case Intrinsic::riscv_masked_atomicrmw_add:
927 case Intrinsic::riscv_masked_atomicrmw_sub:
928 case Intrinsic::riscv_masked_atomicrmw_xchg:
929 case Intrinsic::riscv_masked_atomicrmw_max:
930 case Intrinsic::riscv_masked_atomicrmw_min:
931 case Intrinsic::riscv_masked_atomicrmw_umax:
932 case Intrinsic::riscv_masked_atomicrmw_umin:
933 case Intrinsic::riscv_masked_cmpxchg:
934 return true;
935 }
936}
937
938bool RISCVLegalizerInfo::legalizeVAStart(MachineInstr &MI,
939 MachineIRBuilder &MIRBuilder) const {
940 // Stores the address of the VarArgsFrameIndex slot into the memory location
941 assert(MI.getOpcode() == TargetOpcode::G_VASTART);
942 MachineFunction *MF = MI.getParent()->getParent();
944 int FI = FuncInfo->getVarArgsFrameIndex();
945 LLT AddrTy = MIRBuilder.getMRI()->getType(MI.getOperand(0).getReg());
946 auto FINAddr = MIRBuilder.buildFrameIndex(AddrTy, FI);
947 assert(MI.hasOneMemOperand());
948 MIRBuilder.buildStore(FINAddr, MI.getOperand(0).getReg(),
949 *MI.memoperands()[0]);
950 MI.eraseFromParent();
951 return true;
952}
953
954bool RISCVLegalizerInfo::legalizeReadCounter(
955 MachineInstr &MI, MachineIRBuilder &MIRBuilder,
956 GISelChangeObserver &Observer) const {
957 assert((MI.getOpcode() == TargetOpcode::G_READCYCLECOUNTER ||
958 MI.getOpcode() == TargetOpcode::G_READSTEADYCOUNTER) &&
959 "Unexpected opcode");
960 assert(!STI.is64Bit() && "READCYCLECOUNTER/READSTEADYCOUNTER only "
961 "has custom type legalization on riscv32");
962
963 // On RV32 a 64-bit counter CSR must be read as two 32-bit halves. Because
964 // the count may wrap between the two reads, re-read the high half and loop
965 // until the two high reads agree.
966 int64_t LoCounter, HiCounter;
967 if (MI.getOpcode() == TargetOpcode::G_READCYCLECOUNTER) {
968 LoCounter = RISCVSysReg::cycle;
969 HiCounter = RISCVSysReg::cycleh;
970 } else {
971 LoCounter = RISCVSysReg::time;
972 HiCounter = RISCVSysReg::timeh;
973 }
974
975 MachineBasicBlock *BB = MI.getParent();
976 MachineFunction &MF = *BB->getParent();
977 const BasicBlock *LLVMBB = BB->getBasicBlock();
978 DebugLoc DL = MI.getDebugLoc();
979 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
980
981 // Split BB into an entry that falls through into a loop block, and a done
982 // block that receives the remainder of BB and its original successors.
983 MachineFunction::iterator It = std::next(BB->getIterator());
984 MachineBasicBlock *LoopMBB = MF.CreateMachineBasicBlock(LLVMBB);
985 MachineBasicBlock *DoneMBB = MF.CreateMachineBasicBlock(LLVMBB);
986 MF.insert(It, LoopMBB);
987 MF.insert(It, DoneMBB);
988
989 // Splice the instructions after the readcyclecounter into DoneMBB, notifying
990 // the observer about each moved instruction so CSEInfo stays consistent.
991 for (MachineBasicBlock::iterator I = std::next(MI.getIterator()),
992 E = BB->end();
993 I != E; ++I)
994 Observer.changingInstr(*I);
995 DoneMBB->splice(DoneMBB->begin(), BB,
996 std::next(MachineBasicBlock::iterator(MI)), BB->end());
997 for (MachineInstr &MovedMI : DoneMBB->instrs())
998 Observer.changedInstr(MovedMI);
1000 BB->addSuccessor(LoopMBB);
1001
1002 LLT S32 = LLT::scalar(32);
1003 // Generic vregs carry the s32 type for G_MERGE_VALUES below, but are also
1004 // constrained to GPR so the target CSRRS/BNE instructions satisfy the
1005 // verifier's register-class constraints.
1006 auto CreateGPR = [&]() {
1008 MRI.setRegClass(R, &RISCV::GPRRegClass);
1009 return R;
1010 };
1011 Register LoReg = CreateGPR();
1012 Register HiReg = CreateGPR();
1013 Register ReadAgainReg = CreateGPR();
1014
1015 // read:
1016 // csrrs HiReg, counterh # high word
1017 // csrrs LoReg, counter # low word
1018 // csrrs ReadAgainReg, counterh
1019 // bne HiReg, ReadAgainReg, read
1020 // Emit the target instructions directly with BuildMI.
1021 const RISCVInstrInfo *TII = STI.getInstrInfo();
1022 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), HiReg)
1023 .addImm(HiCounter)
1024 .addReg(RISCV::X0);
1025 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), LoReg)
1026 .addImm(LoCounter)
1027 .addReg(RISCV::X0);
1028 BuildMI(LoopMBB, DL, TII->get(RISCV::CSRRS), ReadAgainReg)
1029 .addImm(HiCounter)
1030 .addReg(RISCV::X0);
1031
1032 BuildMI(LoopMBB, DL, TII->get(RISCV::BNE))
1033 .addReg(HiReg)
1034 .addReg(ReadAgainReg)
1035 .addMBB(LoopMBB);
1036
1037 LoopMBB->addSuccessor(LoopMBB);
1038 LoopMBB->addSuccessor(DoneMBB);
1039
1040 // Re-pair the two halves into the 64-bit result.
1041 Register DstReg = MI.getOperand(0).getReg();
1042 Observer.erasingInstr(MI);
1043 MI.eraseFromParent();
1044
1045 MIRBuilder.setInsertPt(*DoneMBB, DoneMBB->begin());
1046 MIRBuilder.setDebugLoc(DL);
1047 MIRBuilder.buildMergeValues(DstReg, {LoReg, HiReg});
1048 return true;
1049}
1050
1051bool RISCVLegalizerInfo::legalizeBRJT(MachineInstr &MI,
1052 MachineIRBuilder &MIRBuilder) const {
1053 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
1054 auto &MF = *MI.getParent()->getParent();
1055 const MachineJumpTableInfo *MJTI = MF.getJumpTableInfo();
1056 unsigned EntrySize = MJTI->getEntrySize(MF.getDataLayout());
1057
1058 Register PtrReg = MI.getOperand(0).getReg();
1059 LLT PtrTy = MRI.getType(PtrReg);
1060 Register IndexReg = MI.getOperand(2).getReg();
1061 LLT IndexTy = MRI.getType(IndexReg);
1062
1063 if (!isPowerOf2_32(EntrySize))
1064 return false;
1065
1066 auto ShiftAmt = MIRBuilder.buildConstant(IndexTy, Log2_32(EntrySize));
1067 IndexReg = MIRBuilder.buildShl(IndexTy, IndexReg, ShiftAmt).getReg(0);
1068
1069 auto Addr = MIRBuilder.buildPtrAdd(PtrTy, PtrReg, IndexReg);
1070
1071 MachineMemOperand *MMO = MF.getMachineMemOperand(
1073 EntrySize, Align(MJTI->getEntryAlignment(MF.getDataLayout())));
1074
1075 Register TargetReg;
1076 switch (MJTI->getEntryKind()) {
1077 default:
1078 return false;
1080 // For PIC, the sequence is:
1081 // BRIND(load(Jumptable + index) + RelocBase)
1082 // RelocBase can be JumpTable, GOT or some sort of global base.
1083 unsigned LoadOpc =
1084 STI.is64Bit() ? TargetOpcode::G_SEXTLOAD : TargetOpcode::G_LOAD;
1085 auto Load = MIRBuilder.buildLoadInstr(LoadOpc, IndexTy, Addr, *MMO);
1086 TargetReg = MIRBuilder.buildPtrAdd(PtrTy, PtrReg, Load).getReg(0);
1087 break;
1088 }
1090 auto Load = MIRBuilder.buildLoadInstr(TargetOpcode::G_SEXTLOAD, IndexTy,
1091 Addr, *MMO);
1092 TargetReg = MIRBuilder.buildIntToPtr(PtrTy, Load).getReg(0);
1093 break;
1094 }
1096 TargetReg = MIRBuilder.buildLoad(PtrTy, Addr, *MMO).getReg(0);
1097 break;
1098 }
1099
1100 MIRBuilder.buildBrIndirect(TargetReg);
1101
1102 MI.eraseFromParent();
1103 return true;
1104}
1105
1106bool RISCVLegalizerInfo::shouldBeInConstantPool(const APInt &APImm,
1107 bool ShouldOptForSize) const {
1108 assert(APImm.getBitWidth() == 32 || APImm.getBitWidth() == 64);
1109 int64_t Imm = APImm.getSExtValue();
1110 // All simm32 constants should be handled by isel.
1111 // NOTE: The getMaxBuildIntsCost call below should return a value >= 2 making
1112 // this check redundant, but small immediates are common so this check
1113 // should have better compile time.
1114 if (isInt<32>(Imm))
1115 return false;
1116
1117 // We only need to cost the immediate, if constant pool lowering is enabled.
1118 if (!STI.useConstantPoolForLargeInts())
1119 return false;
1120
1122 if (Seq.size() <= STI.getMaxBuildIntsCost())
1123 return false;
1124
1125 // Optimizations below are disabled for opt size. If we're optimizing for
1126 // size, use a constant pool.
1127 if (ShouldOptForSize)
1128 return true;
1129 //
1130 // Special case. See if we can build the constant as (ADD (SLLI X, C), X) do
1131 // that if it will avoid a constant pool.
1132 // It will require an extra temporary register though.
1133 // If we have Zba we can use (ADD_UW X, (SLLI X, 32)) to handle cases where
1134 // low and high 32 bits are the same and bit 31 and 63 are set.
1135 unsigned ShiftAmt, AddOpc;
1136 RISCVMatInt::InstSeq SeqLo =
1137 RISCVMatInt::generateTwoRegInstSeq(Imm, STI, ShiftAmt, AddOpc);
1138 return !(!SeqLo.empty() && (SeqLo.size() + 2) <= STI.getMaxBuildIntsCost());
1139}
1140
1141bool RISCVLegalizerInfo::legalizeVScale(MachineInstr &MI,
1142 MachineIRBuilder &MIB) const {
1143 Register Dst = MI.getOperand(0).getReg();
1144
1145 // We define our scalable vector types for lmul=1 to use a 64 bit known
1146 // minimum size. e.g. <vscale x 2 x i32>. VLENB is in bytes so we calculate
1147 // vscale as VLENB / 8.
1148 static_assert(RISCV::RVVBitsPerBlock == 64, "Unexpected bits per block!");
1149 if (STI.getRealMinVLen() < RISCV::RVVBitsPerBlock)
1150 // Support for VLEN==32 is incomplete.
1151 return false;
1152
1153 // We assume VLENB is a multiple of 8. We manually choose the best shift
1154 // here because SimplifyDemandedBits isn't always able to simplify it.
1155 uint64_t Val = MI.getOperand(1).getCImm()->getZExtValue();
1156 if (isPowerOf2_64(Val)) {
1157 uint64_t Log2 = Log2_64(Val);
1158 if (Log2 < 3) {
1159 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1160 MIB.buildLShr(Dst, VLENB, MIB.buildConstant(sXLen, 3 - Log2),
1162 } else if (Log2 > 3) {
1163 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1164 MIB.buildShl(Dst, VLENB, MIB.buildConstant(sXLen, Log2 - 3));
1165 } else {
1166 MIB.buildInstr(RISCV::G_READ_VLENB, {Dst}, {});
1167 }
1168 } else if ((Val % 8) == 0) {
1169 // If the multiplier is a multiple of 8, scale it down to avoid needing
1170 // to shift the VLENB value.
1171 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1172 MIB.buildMul(Dst, VLENB, MIB.buildConstant(sXLen, Val / 8));
1173 } else {
1174 auto VLENB = MIB.buildInstr(RISCV::G_READ_VLENB, {sXLen}, {});
1175 auto VScale = MIB.buildLShr(sXLen, VLENB, MIB.buildConstant(sXLen, 3),
1177 MIB.buildMul(Dst, VScale, MIB.buildConstant(sXLen, Val));
1178 }
1179 MI.eraseFromParent();
1180 return true;
1181}
1182
1183// Custom-lower extensions from mask vectors by using a vselect either with 1
1184// for zero/any-extension or -1 for sign-extension:
1185// (vXiN = (s|z)ext vXi1:vmask) -> (vXiN = vselect vmask, (-1 or 1), 0)
1186// Note that any-extension is lowered identically to zero-extension.
1187bool RISCVLegalizerInfo::legalizeExt(MachineInstr &MI,
1188 MachineIRBuilder &MIB) const {
1189
1190 unsigned Opc = MI.getOpcode();
1191 assert(Opc == TargetOpcode::G_ZEXT || Opc == TargetOpcode::G_SEXT ||
1192 Opc == TargetOpcode::G_ANYEXT);
1193
1194 MachineRegisterInfo &MRI = *MIB.getMRI();
1195 Register Dst = MI.getOperand(0).getReg();
1196 Register Src = MI.getOperand(1).getReg();
1197
1198 LLT DstTy = MRI.getType(Dst);
1199 int64_t ExtTrueVal = Opc == TargetOpcode::G_SEXT ? -1 : 1;
1200 LLT DstEltTy = DstTy.getElementType();
1201 auto SplatZero = MIB.buildSplatVector(DstTy, MIB.buildConstant(DstEltTy, 0));
1202 auto SplatTrue =
1203 MIB.buildSplatVector(DstTy, MIB.buildConstant(DstEltTy, ExtTrueVal));
1204 MIB.buildSelect(Dst, Src, SplatTrue, SplatZero);
1205
1206 MI.eraseFromParent();
1207 return true;
1208}
1209
1210bool RISCVLegalizerInfo::legalizeLoadStore(MachineInstr &MI,
1211 LegalizerHelper &Helper,
1212 MachineIRBuilder &MIB) const {
1214 "Machine instructions must be Load/Store.");
1215 MachineRegisterInfo &MRI = *MIB.getMRI();
1216 MachineFunction *MF = MI.getMF();
1217 const DataLayout &DL = MIB.getDataLayout();
1218 LLVMContext &Ctx = MF->getFunction().getContext();
1219
1220 Register DstReg = MI.getOperand(0).getReg();
1221 LLT DataTy = MRI.getType(DstReg);
1222 if (!DataTy.isVector())
1223 return false;
1224
1225 if (!MI.hasOneMemOperand())
1226 return false;
1227
1228 MachineMemOperand *MMO = *MI.memoperands_begin();
1229
1230 const auto *TLI = STI.getTargetLowering();
1231 EVT VT = EVT::getEVT(getTypeForLLT(DataTy, Ctx));
1232
1233 if (TLI->allowsMemoryAccessForAlignment(Ctx, DL, VT, *MMO))
1234 return true;
1235
1236 unsigned EltSizeBits = DataTy.getScalarSizeInBits();
1237 assert((EltSizeBits == 16 || EltSizeBits == 32 || EltSizeBits == 64) &&
1238 "Unexpected unaligned RVV load type");
1239
1240 // Calculate the new vector type with i8 elements
1241 unsigned NumElements =
1242 DataTy.getElementCount().getKnownMinValue() * (EltSizeBits / 8);
1243 LLT NewDataTy = LLT::scalable_vector(NumElements, 8);
1244
1245 Helper.bitcast(MI, 0, NewDataTy);
1246
1247 return true;
1248}
1249
1250/// Return the type of the mask type suitable for masking the provided
1251/// vector type. This is simply an i1 element type vector of the same
1252/// (possibly scalable) length.
1253static LLT getMaskTypeFor(LLT VecTy) {
1254 assert(VecTy.isVector());
1255 ElementCount EC = VecTy.getElementCount();
1256 return LLT::vector(EC, LLT::scalar(1));
1257}
1258
1259/// Creates an all ones mask suitable for masking a vector of type VecTy with
1260/// vector length VL.
1262 MachineIRBuilder &MIB,
1263 MachineRegisterInfo &MRI) {
1264 LLT MaskTy = getMaskTypeFor(VecTy);
1265 return MIB.buildInstr(RISCV::G_VMSET_VL, {MaskTy}, {VL});
1266}
1267
1268/// Gets the two common "VL" operands: an all-ones mask and the vector length.
1269/// VecTy is a scalable vector type.
1270static std::pair<MachineInstrBuilder, MachineInstrBuilder>
1272 assert(VecTy.isScalableVector() && "Expecting scalable container type");
1273 const RISCVSubtarget &STI = MIB.getMF().getSubtarget<RISCVSubtarget>();
1274 LLT XLenTy(STI.getXLenVT());
1275 auto VL = MIB.buildConstant(XLenTy, -1);
1276 auto Mask = buildAllOnesMask(VecTy, VL, MIB, MRI);
1277 return {Mask, VL};
1278}
1279
1281buildSplatPartsS64WithVL(const DstOp &Dst, const SrcOp &Passthru, Register Lo,
1282 Register Hi, const SrcOp &VL, MachineIRBuilder &MIB,
1283 MachineRegisterInfo &MRI) {
1284 // TODO: If the Hi bits of the splat are undefined, then it's fine to just
1285 // splat Lo even if it might be sign extended. I don't think we have
1286 // introduced a case where we're build a s64 where the upper bits are undef
1287 // yet.
1288
1289 // Fall back to a stack store and stride x0 vector load.
1290 // TODO: need to lower G_SPLAT_VECTOR_SPLIT_I64. This is done in
1291 // preprocessDAG in SDAG.
1292 return MIB.buildInstr(RISCV::G_SPLAT_VECTOR_SPLIT_I64_VL, {Dst},
1293 {Passthru, Lo, Hi, VL});
1294}
1295
1297buildSplatSplitS64WithVL(const DstOp &Dst, const SrcOp &Passthru,
1298 const SrcOp &Scalar, const SrcOp &VL,
1300 assert(Scalar.getLLTTy(MRI) == LLT::scalar(64) && "Unexpected VecTy!");
1301 auto Unmerge = MIB.buildUnmerge(LLT::scalar(32), Scalar);
1302 return buildSplatPartsS64WithVL(Dst, Passthru, Unmerge.getReg(0),
1303 Unmerge.getReg(1), VL, MIB, MRI);
1304}
1305
1306// Lower splats of s1 types to G_ICMP. For each mask vector type, we have a
1307// legal equivalently-sized i8 type, so we can use that as a go-between.
1308// Splats of s1 types that have constant value can be legalized as VMSET_VL or
1309// VMCLR_VL.
1310bool RISCVLegalizerInfo::legalizeSplatVector(MachineInstr &MI,
1311 MachineIRBuilder &MIB) const {
1312 assert(MI.getOpcode() == TargetOpcode::G_SPLAT_VECTOR);
1313
1314 MachineRegisterInfo &MRI = *MIB.getMRI();
1315
1316 Register Dst = MI.getOperand(0).getReg();
1317 Register SplatVal = MI.getOperand(1).getReg();
1318
1319 LLT VecTy = MRI.getType(Dst);
1320 LLT XLenTy(STI.getXLenVT());
1321
1322 // Handle case of s64 element vectors on rv32
1323 if (XLenTy.getSizeInBits() == 32 &&
1324 VecTy.getElementType().getSizeInBits() == 64) {
1325 auto [_, VL] = buildDefaultVLOps(MRI.getType(Dst), MIB, MRI);
1326 buildSplatSplitS64WithVL(Dst, MIB.buildUndef(VecTy), SplatVal, VL, MIB,
1327 MRI);
1328 MI.eraseFromParent();
1329 return true;
1330 }
1331
1332 // All-zeros or all-ones splats are handled specially.
1333 MachineInstr &SplatValMI = *MRI.getVRegDef(SplatVal);
1334 if (isAllOnesOrAllOnesSplat(SplatValMI, MRI)) {
1335 auto VL = buildDefaultVLOps(VecTy, MIB, MRI).second;
1336 MIB.buildInstr(RISCV::G_VMSET_VL, {Dst}, {VL});
1337 MI.eraseFromParent();
1338 return true;
1339 }
1340 if (isNullOrNullSplat(SplatValMI, MRI)) {
1341 auto VL = buildDefaultVLOps(VecTy, MIB, MRI).second;
1342 MIB.buildInstr(RISCV::G_VMCLR_VL, {Dst}, {VL});
1343 MI.eraseFromParent();
1344 return true;
1345 }
1346
1347 // Handle non-constant mask splat (i.e. not sure if it's all zeros or all
1348 // ones) by promoting it to an s8 splat.
1349 LLT InterEltTy = LLT::scalar(8);
1350 LLT InterTy = VecTy.changeElementType(InterEltTy);
1351 auto ZExtSplatVal = MIB.buildZExt(InterEltTy, SplatVal);
1352 auto And =
1353 MIB.buildAnd(InterEltTy, ZExtSplatVal, MIB.buildConstant(InterEltTy, 1));
1354 auto LHS = MIB.buildSplatVector(InterTy, And);
1355 auto ZeroSplat =
1356 MIB.buildSplatVector(InterTy, MIB.buildConstant(InterEltTy, 0));
1357 MIB.buildICmp(CmpInst::Predicate::ICMP_NE, Dst, LHS, ZeroSplat);
1358 MI.eraseFromParent();
1359 return true;
1360}
1361
1362static LLT getLMUL1Ty(LLT VecTy) {
1363 assert(VecTy.getElementType().getSizeInBits() <= 64 &&
1364 "Unexpected vector LLT");
1366 VecTy.getElementType().getSizeInBits(),
1367 VecTy.getElementType());
1368}
1369
1370bool RISCVLegalizerInfo::legalizeExtractSubvector(MachineInstr &MI,
1371 MachineIRBuilder &MIB) const {
1372 GExtractSubvector &ES = cast<GExtractSubvector>(MI);
1373
1374 MachineRegisterInfo &MRI = *MIB.getMRI();
1375
1376 Register Dst = ES.getReg(0);
1377 Register Src = ES.getSrcVec();
1378 uint64_t Idx = ES.getIndexImm();
1379
1380 // With an index of 0 this is a cast-like subvector, which can be performed
1381 // with subregister operations.
1382 if (Idx == 0)
1383 return true;
1384
1385 LLT LitTy = MRI.getType(Dst);
1386 LLT BigTy = MRI.getType(Src);
1387
1388 if (LitTy.getElementType() == LLT::scalar(1)) {
1389 // We can't slide this mask vector up indexed by its i1 elements.
1390 // This poses a problem when we wish to insert a scalable vector which
1391 // can't be re-expressed as a larger type. Just choose the slow path and
1392 // extend to a larger type, then truncate back down.
1393 LLT ExtBigTy = BigTy.changeElementType(LLT::scalar(8));
1394 LLT ExtLitTy = LitTy.changeElementType(LLT::scalar(8));
1395 auto BigZExt = MIB.buildZExt(ExtBigTy, Src);
1396 auto ExtractZExt = MIB.buildExtractSubvector(ExtLitTy, BigZExt, Idx);
1397 auto SplatZero = MIB.buildSplatVector(
1398 ExtLitTy, MIB.buildConstant(ExtLitTy.getElementType(), 0));
1399 MIB.buildICmp(CmpInst::Predicate::ICMP_NE, Dst, ExtractZExt, SplatZero);
1400 MI.eraseFromParent();
1401 return true;
1402 }
1403
1404 // extract_subvector scales the index by vscale if the subvector is scalable,
1405 // and decomposeSubvectorInsertExtractToSubRegs takes this into account.
1406 const RISCVRegisterInfo *TRI = STI.getRegisterInfo();
1407 MVT LitTyMVT = getMVTForLLT(LitTy);
1408 auto Decompose =
1410 getMVTForLLT(BigTy), LitTyMVT, Idx, TRI);
1411 unsigned RemIdx = Decompose.second;
1412
1413 // If the Idx has been completely eliminated then this is a subvector extract
1414 // which naturally aligns to a vector register. These can easily be handled
1415 // using subregister manipulation.
1416 if (RemIdx == 0)
1417 return true;
1418
1419 // Else LitTy is M1 or smaller and may need to be slid down: if LitTy
1420 // was > M1 then the index would need to be a multiple of VLMAX, and so would
1421 // divide exactly.
1422 assert(
1425
1426 // If the vector type is an LMUL-group type, extract a subvector equal to the
1427 // nearest full vector register type.
1428 LLT InterLitTy = BigTy;
1429 Register Vec = Src;
1431 getLMUL1Ty(BigTy).getSizeInBits())) {
1432 // If BigTy has an LMUL > 1, then LitTy should have a smaller LMUL, and
1433 // we should have successfully decomposed the extract into a subregister.
1434 assert(Decompose.first != RISCV::NoSubRegister);
1435 InterLitTy = getLMUL1Ty(BigTy);
1436 // SDAG builds a TargetExtractSubreg. We cannot create a a Copy with SubReg
1437 // specified on the source Register (the equivalent) since generic virtual
1438 // register does not allow subregister index.
1439 Vec = MIB.buildExtractSubvector(InterLitTy, Src, Idx - RemIdx).getReg(0);
1440 }
1441
1442 // Slide this vector register down by the desired number of elements in order
1443 // to place the desired subvector starting at element 0.
1444 const LLT XLenTy(STI.getXLenVT());
1445 auto SlidedownAmt = MIB.buildVScale(XLenTy, RemIdx);
1446 auto [Mask, VL] = buildDefaultVLOps(InterLitTy, MIB, MRI);
1448 auto Slidedown = MIB.buildInstr(
1449 RISCV::G_VSLIDEDOWN_VL, {InterLitTy},
1450 {MIB.buildUndef(InterLitTy), Vec, SlidedownAmt, Mask, VL, Policy});
1451
1452 // Now the vector is in the right position, extract our final subvector. This
1453 // should resolve to a COPY.
1454 MIB.buildExtractSubvector(Dst, Slidedown, 0);
1455
1456 MI.eraseFromParent();
1457 return true;
1458}
1459
1460bool RISCVLegalizerInfo::legalizeInsertSubvector(MachineInstr &MI,
1461 LegalizerHelper &Helper,
1462 MachineIRBuilder &MIB) const {
1463 GInsertSubvector &IS = cast<GInsertSubvector>(MI);
1464
1465 MachineRegisterInfo &MRI = *MIB.getMRI();
1466
1467 Register Dst = IS.getReg(0);
1468 Register BigVec = IS.getBigVec();
1469 Register LitVec = IS.getSubVec();
1470 uint64_t Idx = IS.getIndexImm();
1471
1472 LLT BigTy = MRI.getType(BigVec);
1473 LLT LitTy = MRI.getType(LitVec);
1474
1475 if (Idx == 0 && mi_match(BigVec, MRI, m_GImplicitDef()))
1476 return true;
1477
1478 // We don't have the ability to slide mask vectors up indexed by their i1
1479 // elements; the smallest we can do is i8. Often we are able to bitcast to
1480 // equivalent i8 vectors. Otherwise, we can must zeroextend to equivalent i8
1481 // vectors and truncate down after the insert.
1482 if (LitTy.getElementType() == LLT::scalar(1)) {
1483 auto BigTyMinElts = BigTy.getElementCount().getKnownMinValue();
1484 auto LitTyMinElts = LitTy.getElementCount().getKnownMinValue();
1485 if (BigTyMinElts >= 8 && LitTyMinElts >= 8)
1486 return Helper.bitcast(
1487 IS, 0,
1489
1490 // We can't slide this mask vector up indexed by its i1 elements.
1491 // This poses a problem when we wish to insert a scalable vector which
1492 // can't be re-expressed as a larger type. Just choose the slow path and
1493 // extend to a larger type, then truncate back down.
1494 LLT ExtBigTy = BigTy.changeElementType(LLT::scalar(8));
1495 return Helper.widenScalar(IS, 0, ExtBigTy);
1496 }
1497
1498 const RISCVRegisterInfo *TRI = STI.getRegisterInfo();
1499 unsigned SubRegIdx, RemIdx;
1500 std::tie(SubRegIdx, RemIdx) =
1502 getMVTForLLT(BigTy), getMVTForLLT(LitTy), Idx, TRI);
1503
1504 TypeSize VecRegSize = TypeSize::getScalable(RISCV::RVVBitsPerBlock);
1506 STI.expandVScale(LitTy.getSizeInBits()).getKnownMinValue()));
1507 bool ExactlyVecRegSized =
1508 STI.expandVScale(LitTy.getSizeInBits())
1509 .isKnownMultipleOf(STI.expandVScale(VecRegSize));
1510
1511 // If the Idx has been completely eliminated and this subvector's size is a
1512 // vector register or a multiple thereof, or the surrounding elements are
1513 // undef, then this is a subvector insert which naturally aligns to a vector
1514 // register. These can easily be handled using subregister manipulation.
1515 if (RemIdx == 0 && ExactlyVecRegSized)
1516 return true;
1517
1518 // If the subvector is smaller than a vector register, then the insertion
1519 // must preserve the undisturbed elements of the register. We do this by
1520 // lowering to an EXTRACT_SUBVECTOR grabbing the nearest LMUL=1 vector type
1521 // (which resolves to a subregister copy), performing a VSLIDEUP to place the
1522 // subvector within the vector register, and an INSERT_SUBVECTOR of that
1523 // LMUL=1 type back into the larger vector (resolving to another subregister
1524 // operation). See below for how our VSLIDEUP works. We go via a LMUL=1 type
1525 // to avoid allocating a large register group to hold our subvector.
1526
1527 // VSLIDEUP works by leaving elements 0<i<OFFSET undisturbed, elements
1528 // OFFSET<=i<VL set to the "subvector" and vl<=i<VLMAX set to the tail policy
1529 // (in our case undisturbed). This means we can set up a subvector insertion
1530 // where OFFSET is the insertion offset, and the VL is the OFFSET plus the
1531 // size of the subvector.
1532 const LLT XLenTy(STI.getXLenVT());
1533 LLT InterLitTy = BigTy;
1534 Register AlignedExtract = BigVec;
1535 unsigned AlignedIdx = Idx - RemIdx;
1537 getLMUL1Ty(BigTy).getSizeInBits())) {
1538 InterLitTy = getLMUL1Ty(BigTy);
1539 // Extract a subvector equal to the nearest full vector register type. This
1540 // should resolve to a G_EXTRACT on a subreg.
1541 AlignedExtract =
1542 MIB.buildExtractSubvector(InterLitTy, BigVec, AlignedIdx).getReg(0);
1543 }
1544
1545 auto Insert = MIB.buildInsertSubvector(InterLitTy, MIB.buildUndef(InterLitTy),
1546 LitVec, 0);
1547
1548 auto [Mask, _] = buildDefaultVLOps(InterLitTy, MIB, MRI);
1549 auto VL = MIB.buildVScale(XLenTy, LitTy.getElementCount().getKnownMinValue());
1550
1551 // If we're inserting into the lowest elements, use a tail undisturbed
1552 // vmv.v.v.
1553 MachineInstrBuilder Inserted;
1554 bool NeedInsertSubvec =
1555 TypeSize::isKnownGT(BigTy.getSizeInBits(), InterLitTy.getSizeInBits());
1556 Register InsertedDst =
1557 NeedInsertSubvec ? MRI.createGenericVirtualRegister(InterLitTy) : Dst;
1558 if (RemIdx == 0) {
1559 Inserted = MIB.buildInstr(RISCV::G_VMV_V_V_VL, {InsertedDst},
1560 {AlignedExtract, Insert, VL});
1561 } else {
1562 auto SlideupAmt = MIB.buildVScale(XLenTy, RemIdx);
1563 // Construct the vector length corresponding to RemIdx + length(LitTy).
1564 VL = MIB.buildAdd(XLenTy, SlideupAmt, VL);
1565 // Use tail agnostic policy if we're inserting over InterLitTy's tail.
1566 ElementCount EndIndex =
1569 if (STI.expandVScale(EndIndex) ==
1570 STI.expandVScale(InterLitTy.getElementCount()))
1572
1573 Inserted =
1574 MIB.buildInstr(RISCV::G_VSLIDEUP_VL, {InsertedDst},
1575 {AlignedExtract, Insert, SlideupAmt, Mask, VL, Policy});
1576 }
1577
1578 // If required, insert this subvector back into the correct vector register.
1579 // This should resolve to an INSERT_SUBREG instruction.
1580 if (NeedInsertSubvec)
1581 MIB.buildInsertSubvector(Dst, BigVec, Inserted, AlignedIdx);
1582
1583 MI.eraseFromParent();
1584 return true;
1585}
1586
1587bool RISCVLegalizerInfo::legalizeBitreverse(MachineInstr &MI,
1588 MachineIRBuilder &MIB) const {
1589 assert(MI.getOpcode() == TargetOpcode::G_BITREVERSE && "Unexpected opcode");
1590
1591 if (!STI.hasStdExtZbkb())
1592 return false;
1593
1594 MachineRegisterInfo &MRI = *MIB.getMRI();
1595
1596 Register Dst = MI.getOperand(0).getReg();
1597 Register Src = MI.getOperand(1).getReg();
1598
1599 if (!MRI.getType(Dst).isScalar(8))
1600 return false;
1601
1602 auto WideSrc = MIB.buildAnyExt(sXLen, Src);
1603 auto Brev = MIB.buildInstr(RISCV::G_BREV8, {sXLen}, {WideSrc.getReg(0)});
1604 MIB.buildTrunc(Dst, Brev.getReg(0));
1605
1606 MI.eraseFromParent();
1607 return true;
1608}
1609
1610static unsigned getRISCVWOpcode(unsigned Opcode) {
1611 switch (Opcode) {
1612 default:
1613 llvm_unreachable("Unexpected opcode");
1614 case TargetOpcode::G_ASHR:
1615 return RISCV::G_SRAW;
1616 case TargetOpcode::G_LSHR:
1617 return RISCV::G_SRLW;
1618 case TargetOpcode::G_SHL:
1619 return RISCV::G_SLLW;
1620 case TargetOpcode::G_SDIV:
1621 return RISCV::G_DIVW;
1622 case TargetOpcode::G_UDIV:
1623 return RISCV::G_DIVUW;
1624 case TargetOpcode::G_UREM:
1625 return RISCV::G_REMUW;
1626 case TargetOpcode::G_ROTL:
1627 return RISCV::G_ROLW;
1628 case TargetOpcode::G_ROTR:
1629 return RISCV::G_RORW;
1630 case TargetOpcode::G_CTLZ:
1631 return RISCV::G_CLZW;
1632 case TargetOpcode::G_CTTZ:
1633 return RISCV::G_CTZW;
1634 case TargetOpcode::G_CTLS:
1635 return RISCV::G_CLSW;
1636 case TargetOpcode::G_FPTOSI:
1637 return RISCV::G_FCVT_W_RV64;
1638 case TargetOpcode::G_FPTOUI:
1639 return RISCV::G_FCVT_WU_RV64;
1640 }
1641}
1642
1645 LostDebugLocObserver &LocObserver) const {
1646 MachineIRBuilder &MIRBuilder = Helper.MIRBuilder;
1647 MachineRegisterInfo &MRI = *MIRBuilder.getMRI();
1648 MachineFunction &MF = *MI.getParent()->getParent();
1649 switch (MI.getOpcode()) {
1650 default:
1651 // No idea what to do.
1652 return false;
1653 case TargetOpcode::G_ABS:
1654 return Helper.lowerAbsToMaxNeg(MI);
1655 case TargetOpcode::G_CLMULH:
1656 case TargetOpcode::G_CLMULR: {
1657 assert(STI.is64Bit() &&
1658 MRI.getType(MI.getOperand(0).getReg()) == LLT::scalar(32) &&
1659 "Unexpected custom legalization");
1660 // Shift both inputs by 32 so the full product has 64 trailing zeros.
1661 // Perform CLMULH or CLMULR on the shifted inputs, then extract the upper
1662 // 32 bits of the result.
1663 auto Shift = MIRBuilder.buildConstant(sXLen, 32);
1664 auto LHS = MIRBuilder.buildAnyExt(sXLen, MI.getOperand(1));
1665 auto RHS = MIRBuilder.buildAnyExt(sXLen, MI.getOperand(2));
1666 auto ShiftedLHS = MIRBuilder.buildShl(sXLen, LHS, Shift);
1667 auto ShiftedRHS = MIRBuilder.buildShl(sXLen, RHS, Shift);
1668 auto Product = MIRBuilder.buildInstr(MI.getOpcode(), {sXLen},
1669 {ShiftedLHS, ShiftedRHS});
1670 auto High = MIRBuilder.buildLShr(sXLen, Product, Shift);
1671 MIRBuilder.buildTrunc(MI.getOperand(0), High);
1672 MI.eraseFromParent();
1673 return true;
1674 }
1675 case TargetOpcode::G_FCONSTANT: {
1676 const APFloat &FVal = MI.getOperand(1).getFPImm()->getValueAPF();
1677
1678 // Convert G_FCONSTANT to G_CONSTANT.
1679 Register DstReg = MI.getOperand(0).getReg();
1680 MIRBuilder.buildConstant(DstReg, FVal.bitcastToAPInt());
1681
1682 MI.eraseFromParent();
1683 return true;
1684 }
1685 case TargetOpcode::G_CONSTANT: {
1686 const Function &F = MF.getFunction();
1687 // TODO: if PSI and BFI are present, add " ||
1688 // llvm::shouldOptForSize(*CurMBB, PSI, BFI)".
1689 bool ShouldOptForSize = F.hasOptSize();
1690 const ConstantInt *ConstVal = MI.getOperand(1).getCImm();
1691 if (!shouldBeInConstantPool(ConstVal->getValue(), ShouldOptForSize))
1692 return true;
1693 return Helper.lowerConstant(MI);
1694 }
1695 case TargetOpcode::G_SUB:
1696 case TargetOpcode::G_ADD: {
1697 Helper.Observer.changingInstr(MI);
1698 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1699 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1700
1701 Register DstALU = MRI.createGenericVirtualRegister(sXLen);
1702
1703 MachineOperand &MO = MI.getOperand(0);
1704 MIRBuilder.setInsertPt(MIRBuilder.getMBB(), ++MIRBuilder.getInsertPt());
1705 auto DstSext = MIRBuilder.buildSExtInReg(sXLen, DstALU, 32);
1706
1707 MIRBuilder.buildInstr(TargetOpcode::G_TRUNC, {MO}, {DstSext});
1708 MO.setReg(DstALU);
1709
1710 Helper.Observer.changedInstr(MI);
1711 return true;
1712 }
1713 case TargetOpcode::G_ASHR:
1714 case TargetOpcode::G_LSHR:
1715 case TargetOpcode::G_SHL: {
1716 if (getIConstantVRegValWithLookThrough(MI.getOperand(2).getReg(), MRI)) {
1717 // We don't need a custom node for shift by constant. Just widen the
1718 // source and the shift amount.
1719 unsigned ExtOpc = TargetOpcode::G_ANYEXT;
1720 if (MI.getOpcode() == TargetOpcode::G_ASHR)
1721 ExtOpc = TargetOpcode::G_SEXT;
1722 else if (MI.getOpcode() == TargetOpcode::G_LSHR)
1723 ExtOpc = TargetOpcode::G_ZEXT;
1724
1725 Helper.Observer.changingInstr(MI);
1726 Helper.widenScalarSrc(MI, sXLen, 1, ExtOpc);
1727 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ZEXT);
1728 Helper.widenScalarDst(MI, sXLen);
1729 Helper.Observer.changedInstr(MI);
1730 return true;
1731 }
1732
1733 Helper.Observer.changingInstr(MI);
1734 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1735 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1736 Helper.widenScalarDst(MI, sXLen);
1737 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1738 Helper.Observer.changedInstr(MI);
1739 return true;
1740 }
1741 case TargetOpcode::G_SDIV:
1742 case TargetOpcode::G_UDIV:
1743 case TargetOpcode::G_UREM:
1744 case TargetOpcode::G_ROTL:
1745 case TargetOpcode::G_ROTR: {
1746 Helper.Observer.changingInstr(MI);
1747 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1748 Helper.widenScalarSrc(MI, sXLen, 2, TargetOpcode::G_ANYEXT);
1749 Helper.widenScalarDst(MI, sXLen);
1750 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1751 Helper.Observer.changedInstr(MI);
1752 return true;
1753 }
1754 case TargetOpcode::G_CTLZ:
1755 case TargetOpcode::G_CTTZ:
1756 case TargetOpcode::G_CTLS: {
1757 Helper.Observer.changingInstr(MI);
1758 Helper.widenScalarSrc(MI, sXLen, 1, TargetOpcode::G_ANYEXT);
1759 Helper.widenScalarDst(MI, sXLen);
1760 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1761 Helper.Observer.changedInstr(MI);
1762 return true;
1763 }
1764 case TargetOpcode::G_FPTOSI:
1765 case TargetOpcode::G_FPTOUI: {
1766 Helper.Observer.changingInstr(MI);
1767 Helper.widenScalarDst(MI, sXLen);
1768 MI.setDesc(MIRBuilder.getTII().get(getRISCVWOpcode(MI.getOpcode())));
1770 Helper.Observer.changedInstr(MI);
1771 return true;
1772 }
1773 case TargetOpcode::G_LROUND: {
1774 // The (i32 any_lround) Pat is IsRV32-only; on RV64 lower to
1775 // riscv_fcvt_w_rv64 with FRM_RMM.
1776 Helper.Observer.changingInstr(MI);
1777 Helper.widenScalarDst(MI, sXLen);
1778 MI.setDesc(MIRBuilder.getTII().get(RISCV::G_FCVT_W_RV64));
1780 Helper.Observer.changedInstr(MI);
1781 return true;
1782 }
1783 case TargetOpcode::G_READCYCLECOUNTER:
1784 case TargetOpcode::G_READSTEADYCOUNTER:
1785 return legalizeReadCounter(MI, MIRBuilder, Helper.Observer);
1786 case TargetOpcode::G_IS_FPCLASS: {
1787 Register GISFPCLASS = MI.getOperand(0).getReg();
1788 Register Src = MI.getOperand(1).getReg();
1789 const MachineOperand &ImmOp = MI.getOperand(2);
1790 MachineIRBuilder MIB(MI);
1791
1792 // Turn LLVM IR's floating point classes to that in RISC-V,
1793 // by simply rotating the 10-bit immediate right by two bits.
1794 APInt GFpClassImm(10, static_cast<uint64_t>(ImmOp.getImm()));
1795 auto FClassMask = MIB.buildConstant(sXLen, GFpClassImm.rotr(2).zext(XLen));
1796 auto ConstZero = MIB.buildConstant(sXLen, 0);
1797
1798 auto GFClass = MIB.buildInstr(RISCV::G_FCLASS, {sXLen}, {Src});
1799 auto And = MIB.buildAnd(sXLen, GFClass, FClassMask);
1800 MIB.buildICmp(CmpInst::ICMP_NE, GISFPCLASS, And, ConstZero);
1801
1802 MI.eraseFromParent();
1803 return true;
1804 }
1805 case TargetOpcode::G_BRJT:
1806 return legalizeBRJT(MI, MIRBuilder);
1807 case TargetOpcode::G_VASTART:
1808 return legalizeVAStart(MI, MIRBuilder);
1809 case TargetOpcode::G_VSCALE:
1810 return legalizeVScale(MI, MIRBuilder);
1811 case TargetOpcode::G_ZEXT:
1812 case TargetOpcode::G_SEXT:
1813 case TargetOpcode::G_ANYEXT:
1814 return legalizeExt(MI, MIRBuilder);
1815 case TargetOpcode::G_SPLAT_VECTOR:
1816 return legalizeSplatVector(MI, MIRBuilder);
1817 case TargetOpcode::G_EXTRACT_SUBVECTOR:
1818 return legalizeExtractSubvector(MI, MIRBuilder);
1819 case TargetOpcode::G_INSERT_SUBVECTOR:
1820 return legalizeInsertSubvector(MI, Helper, MIRBuilder);
1821 case TargetOpcode::G_BITREVERSE:
1822 return legalizeBitreverse(MI, MIRBuilder);
1823 case TargetOpcode::G_LOAD:
1824 case TargetOpcode::G_STORE:
1825 return legalizeLoadStore(MI, Helper, MIRBuilder);
1826 }
1827
1828 llvm_unreachable("expected switch to return");
1829}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
unsigned Imm
unsigned uint64_t
constexpr LLT S32
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
#define _
IRTranslator LLVM IR MI
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
This file declares the MachineConstantPool class which is an abstract constant pool to keep track of ...
This file declares the MachineIRBuilder class.
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
ppc ctr loops verify
static LLT getLMUL1Ty(LLT VecTy)
static MachineInstrBuilder buildAllOnesMask(LLT VecTy, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
Creates an all ones mask suitable for masking a vector of type VecTy with vector length VL.
static std::pair< MachineInstrBuilder, MachineInstrBuilder > buildDefaultVLOps(LLT VecTy, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
Gets the two common "VL" operands: an all-ones mask and the vector length.
static LegalityPredicate typeIsLegalBoolVec(unsigned TypeIdx, std::initializer_list< LLT > BoolVecTys, const RISCVSubtarget &ST)
static MachineInstrBuilder buildSplatSplitS64WithVL(const DstOp &Dst, const SrcOp &Passthru, const SrcOp &Scalar, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
static LegalityPredicate typeIsLegalIntOrFPVec(unsigned TypeIdx, std::initializer_list< LLT > IntOrFPVecTys, const RISCVSubtarget &ST)
static MachineInstrBuilder buildSplatPartsS64WithVL(const DstOp &Dst, const SrcOp &Passthru, Register Lo, Register Hi, const SrcOp &VL, MachineIRBuilder &MIB, MachineRegisterInfo &MRI)
static LLT getMaskTypeFor(LLT VecTy)
Return the type of the mask type suitable for masking the provided vector type.
static LegalityPredicate typeIsLegalPtrVec(unsigned TypeIdx, std::initializer_list< LLT > PtrVecTys, const RISCVSubtarget &ST)
static unsigned getRISCVWOpcode(unsigned Opcode)
This file declares the targeting of the Machinelegalizer class for RISC-V.
Value * LHS
APInt bitcastToAPInt() const
Definition APFloat.h:1475
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
LLVM_ABI APInt rotr(unsigned rotateAmt) const
Rotate right by rotateAmt.
Definition APInt.cpp:1199
int64_t getSExtValue() const
Get sign extended value.
Definition APInt.h:1582
@ ICMP_NE
not equal
Definition InstrTypes.h:762
This is the shared class of boolean and integer constants.
Definition Constants.h:87
const APInt & getValue() const
Return the constant as an APInt value reference.
Definition Constants.h:159
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
static constexpr ElementCount getScalable(ScalarTy MinVal)
Definition TypeSize.h:308
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
Abstract class that contains various methods for clients to notify about changes.
virtual void changingInstr(MachineInstr &MI)=0
This instruction is about to be mutated in some way.
virtual void changedInstr(MachineInstr &MI)=0
This instruction was mutated in some way.
virtual void erasingInstr(MachineInstr &MI)=0
An instruction is about to be erased.
Register getReg(unsigned Idx) const
Access the Idx'th operand as a register and return it.
constexpr bool isScalableVector() const
Returns true if the LLT is a scalable vector.
constexpr unsigned getScalarSizeInBits() const
constexpr bool isScalar() const
static constexpr LLT scalable_vector(unsigned MinNumElements, unsigned ScalarSizeInBits)
Get a low-level scalable vector of some number of elements and element width.
constexpr LLT changeElementType(LLT NewEltTy) const
If this type is a vector, return a vector with the same number of elements but the new element type.
static constexpr LLT vector(ElementCount EC, unsigned ScalarSizeInBits)
Get a low-level vector of some number of elements and element width.
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
constexpr bool isVector() const
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
constexpr ElementCount getElementCount() const
static constexpr LLT float16()
Get a 16-bit IEEE half value.
LLT getElementType() const
Returns the vector's element type. Only valid for vector types.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LegalizeRuleSet & maxScalar(unsigned TypeIdx, const LLT Ty)
Ensure the scalar is at most as wide as Ty.
LegalizeRuleSet & lower()
The instruction is lowered.
LegalizeRuleSet & clampScalar(unsigned TypeIdx, const LLT MinTy, const LLT MaxTy)
Limit the range of scalar sizes to MinTy and MaxTy.
LegalizeRuleSet & alwaysLegal()
LegalizeRuleSet & customIf(LegalityPredicate Predicate)
LegalizeRuleSet & widenScalarToNextPow2(unsigned TypeIdx, unsigned MinSize=0)
Widen the scalar to the next power of two that is at least MinSize.
LegalizeRuleSet & customFor(std::initializer_list< LLT > Types)
LLVM_ABI void widenScalarSrc(MachineInstr &MI, LLT WideTy, unsigned OpIdx, unsigned ExtOpcode)
Legalize a single operand OpIdx of the machine instruction MI as a Use by extending the operand's typ...
LLVM_ABI LegalizeResult lowerAbsToMaxNeg(MachineInstr &MI)
LLVM_ABI LegalizeResult bitcast(MachineInstr &MI, unsigned TypeIdx, LLT Ty)
Legalize an instruction by replacing the value type.
GISelChangeObserver & Observer
To keep track of changes made by the LegalizerHelper.
LLVM_ABI LegalizeResult widenScalar(MachineInstr &MI, unsigned TypeIdx, LLT WideTy)
Legalize an instruction by performing the operation on a wider scalar type (for example a 16-bit addi...
MachineIRBuilder & MIRBuilder
Expose MIRBuilder so clients can set their own RecordInsertInstruction functions.
LLVM_ABI LegalizeResult lowerConstant(MachineInstr &MI)
LLVM_ABI void widenScalarDst(MachineInstr &MI, LLT WideTy, unsigned OpIdx=0, unsigned TruncOpcode=TargetOpcode::G_TRUNC)
Legalize a single operand OpIdx of the machine instruction MI as a Def by extending the operand's typ...
LegalizeRuleSet & getActionDefinitionsBuilder(unsigned Opcode)
Get the action definition builder for the given opcode.
const MCInstrDesc & get(unsigned Opcode) const
Return the machine instruction descriptor that corresponds to the specified instruction opcode.
Definition MCInstrInfo.h:89
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
const BasicBlock * getBasicBlock() const
Return the LLVM basic block that this instance corresponded to originally.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
const MachineJumpTableInfo * getJumpTableInfo() const
getJumpTableInfo - Return the jump table info object for the current function.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
Helper class to build MachineInstr.
void setInsertPt(MachineBasicBlock &MBB, MachineBasicBlock::iterator II)
Set the insertion point before the specified position.
MachineInstrBuilder buildAdd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ADD Op0, Op1.
MachineInstrBuilder buildUndef(const DstOp &Res)
Build and insert Res = IMPLICIT_DEF.
MachineInstrBuilder buildUnmerge(ArrayRef< LLT > Res, const SrcOp &Op)
Build and insert Res0, ... = G_UNMERGE_VALUES Op.
MachineInstrBuilder buildSelect(const DstOp &Res, const SrcOp &Tst, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert a Res = G_SELECT Tst, Op0, Op1.
MachineInstrBuilder buildMul(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_MUL Op0, Op1.
MachineInstrBuilder buildInsertSubvector(const DstOp &Res, const SrcOp &Src0, const SrcOp &Src1, unsigned Index)
Build and insert Res = G_INSERT_SUBVECTOR Src0, Src1, Idx.
MachineInstrBuilder buildAnd(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1)
Build and insert Res = G_AND Op0, Op1.
const TargetInstrInfo & getTII()
MachineInstrBuilder buildICmp(CmpInst::Predicate Pred, const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert a Res = G_ICMP Pred, Op0, Op1.
MachineInstrBuilder buildLShr(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
MachineBasicBlock::iterator getInsertPt()
Current insertion point for new instructions.
MachineInstrBuilder buildZExt(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_ZEXT Op.
MachineInstrBuilder buildVScale(const DstOp &Res, unsigned MinElts)
Build and insert Res = G_VSCALE MinElts.
MachineInstrBuilder buildIntToPtr(const DstOp &Dst, const SrcOp &Src)
Build and insert a G_INTTOPTR instruction.
MachineInstrBuilder buildLoad(const DstOp &Res, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert Res = G_LOAD Addr, MMO.
MachineInstrBuilder buildPtrAdd(const DstOp &Res, const SrcOp &Op0, const SrcOp &Op1, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_PTR_ADD Op0, Op1.
MachineInstrBuilder buildShl(const DstOp &Dst, const SrcOp &Src0, const SrcOp &Src1, std::optional< unsigned > Flags=std::nullopt)
MachineInstrBuilder buildStore(const SrcOp &Val, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert G_STORE Val, Addr, MMO.
MachineInstrBuilder buildInstr(unsigned Opcode)
Build and insert <empty> = Opcode <empty>.
MachineInstrBuilder buildFrameIndex(const DstOp &Res, int Idx)
Build and insert Res = G_FRAME_INDEX Idx.
MachineFunction & getMF()
Getter for the function we currently build.
MachineInstrBuilder buildMergeValues(const DstOp &Res, ArrayRef< Register > Ops)
Build and insert Res = G_MERGE_VALUES Op0, ...
MachineInstrBuilder buildTrunc(const DstOp &Res, const SrcOp &Op, std::optional< unsigned > Flags=std::nullopt)
Build and insert Res = G_TRUNC Op.
void setDebugLoc(const DebugLoc &DL)
Set the debug location to DL for all the next build instructions.
const MachineBasicBlock & getMBB() const
Getter for the basic block we currently build.
MachineInstrBuilder buildAnyExt(const DstOp &Res, const SrcOp &Op)
Build and insert Res = G_ANYEXT Op0.
MachineRegisterInfo * getMRI()
Getter for MRI.
MachineInstrBuilder buildExtractSubvector(const DstOp &Res, const SrcOp &Src, unsigned Index)
Build and insert Res = G_EXTRACT_SUBVECTOR Src, Idx0.
const DataLayout & getDataLayout() const
MachineInstrBuilder buildBrIndirect(Register Tgt)
Build and insert G_BRINDIRECT Tgt.
MachineInstrBuilder buildSplatVector(const DstOp &Res, const SrcOp &Val)
Build and insert Res = G_SPLAT_VECTOR Val.
MachineInstrBuilder buildLoadInstr(unsigned Opcode, const DstOp &Res, const SrcOp &Addr, MachineMemOperand &MMO)
Build and insert Res = <opcode> Addr, MMO.
virtual MachineInstrBuilder buildConstant(const DstOp &Res, const ConstantInt &Val)
Build and insert Res = G_CONSTANT Val.
MachineInstrBuilder buildSExtInReg(const DstOp &Res, const SrcOp &Op, int64_t ImmOp)
Build and insert Res = G_SEXT_INREG Op, ImmOp.
Register getReg(unsigned Idx) const
Get the register for the operand index.
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
Representation of each machine instruction.
LLVM_ABI unsigned getEntrySize(const DataLayout &TD) const
getEntrySize - Return the size of each entry in the jump table.
@ EK_LabelDifference32
EK_LabelDifference32 - Each entry is the address of the block minus the address of the jump table.
@ EK_Custom32
EK_Custom32 - Each entry is a 32-bit value that is custom lowered by the TargetLowering::LowerCustomJ...
@ EK_BlockAddress
EK_BlockAddress - Each entry is a plain address of block, e.g.: .word LBB123.
LLVM_ABI unsigned getEntryAlignment(const DataLayout &TD) const
getEntryAlignment - Return the alignment of each entry in the jump table.
A description of a memory reference used in the backend.
@ MOLoad
The memory access reads data.
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
LLVM_ABI void setReg(Register Reg)
Change the register this operand corresponds to.
static MachineOperand CreateImm(int64_t Val)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register createGenericVirtualRegister(LLT Ty, StringRef Name="")
Create and return a new generic virtual register with low-level type Ty.
bool legalizeCustom(LegalizerHelper &Helper, MachineInstr &MI, LostDebugLocObserver &LocObserver) const override
Called for instructions with the Custom LegalizationAction.
bool legalizeIntrinsic(LegalizerHelper &Helper, MachineInstr &MI) const override
RISCVLegalizerInfo(const RISCVSubtarget &ST)
RISCVMachineFunctionInfo - This class is derived from MachineFunctionInfo and contains private RISCV-...
static std::pair< unsigned, unsigned > decomposeSubvectorInsertExtractToSubRegs(MVT VecVT, MVT SubVecVT, unsigned InsertExtractIdx, const RISCVRegisterInfo *TRI)
static RISCVVType::VLMUL getLMUL(MVT VT)
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Register getReg() const
static constexpr TypeSize getScalable(ScalarTy MinimumSize)
Definition TypeSize.h:342
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
static constexpr bool isKnownGT(const FixedOrScalableQuantity &LHS, const FixedOrScalableQuantity &RHS)
Definition TypeSize.h:223
constexpr LeafTy divideCoefficientBy(ScalarTy RHS) const
We do not provide the '/' operator here because division for polynomial types does not work in the sa...
Definition TypeSize.h:252
self_iterator getIterator()
Definition ilist_node.h:123
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
@ BasicBlock
Various leaf nodes.
Definition ISDOpcodes.h:83
LLVM_ABI LegalityPredicate typeInSet(unsigned TypeIdx, std::initializer_list< LLT > TypesInit)
True iff the given type index is one of the specified types.
LLVM_ABI LegalityPredicate immIs(unsigned ImmIdx, int64_t Imm)
True iff the immediate at the given index has the specified value.
Predicate any(Predicate P0, Predicate P1)
True iff P0 or P1 are true.
LLVM_ABI LegalityPredicate sizeIs(unsigned TypeIdx, unsigned Size)
True if the total bitwidth of the specified type index is Size bits.
Predicate all(Predicate P0, Predicate P1)
True iff P0 and P1 are true.
LLVM_ABI LegalityPredicate typeIs(unsigned TypeIdx, LLT TypesInit)
True iff the given type index is the specified type.
LLVM_ABI LegalityPredicate immInSet(unsigned ImmIdx, std::initializer_list< int64_t > ImmsInit)
True iff the immediate at the given index has one of the specified values.
LLVM_ABI LegalizeMutation changeTo(unsigned TypeIdx, LLT Ty)
Select this specific type for the given type index.
ImplicitDefMatch m_GImplicitDef()
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
InstSeq generateInstSeq(int64_t Val, const MCSubtargetInfo &STI)
InstSeq generateTwoRegInstSeq(int64_t Val, const MCSubtargetInfo &STI, unsigned &ShiftAmt, unsigned &AddOpc)
SmallVector< Inst, 8 > InstSeq
Definition RISCVMatInt.h:43
LLVM_ABI std::pair< unsigned, bool > decodeVLMUL(VLMUL VLMul)
static constexpr unsigned RVVBitsPerBlock
Invariant opcodes: All instruction sets have these as their low opcodes.
This is an optimization pass for GlobalISel generic memory operations.
LLVM_ABI Type * getTypeForLLT(LLT Ty, LLVMContext &C)
Get the type back from LLT.
Definition Utils.cpp:1973
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isAllOnesOrAllOnesSplat(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndefs=false)
Return true if the value is a constant -1 integer or a splatted vector of a constant -1 integer (with...
Definition Utils.cpp:1557
@ Load
The value being inserted comes from a load (InsertElement only).
LLVM_ABI MVT getMVTForLLT(LLT Ty)
Get a rough equivalent of an MVT for a given LLT.
constexpr bool isPowerOf2_64(uint64_t Value)
Return true if the argument is a power of two > 0 (64 bit edition.)
Definition MathExtras.h:285
LLVM_ABI bool isNullOrNullSplat(const MachineInstr &MI, const MachineRegisterInfo &MRI, bool AllowUndefs=false)
Return true if the value is a constant 0 integer or a splatted vector of a constant 0 integer (with n...
Definition Utils.cpp:1539
unsigned Log2_64(uint64_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:332
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
std::function< bool(const LegalityQuery &)> LegalityPredicate
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ And
Bitwise or logical AND of integers.
DWARFExpression::Operation Op
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
static LLVM_ABI EVT getEVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
The LegalityQuery object bundles together all the information that's needed to decide whether a given...
ArrayRef< LLT > Types
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getJumpTable(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a jump table entry.