LLVM 24.0.0git
AMDGPUISelLowering.cpp
Go to the documentation of this file.
1//===-- AMDGPUISelLowering.cpp - AMDGPU Common DAG lowering functions -----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This is the parent TargetLowering class for hardware code gen
11/// targets.
12//
13//===----------------------------------------------------------------------===//
14
15#include "AMDGPUISelLowering.h"
16#include "AMDGPU.h"
17#include "AMDGPUInstrInfo.h"
19#include "AMDGPUMemoryUtils.h"
26#include "llvm/IR/IntrinsicsAMDGPU.h"
30
31using namespace llvm;
32
33#define GET_CALLING_CONV_IMPL
34#include "AMDGPUGenCallingConv.inc"
35
37 "amdgpu-bypass-slow-div",
38 cl::desc("Skip 64-bit divide for dynamic 32-bit values"),
39 cl::init(true));
40
41// Find a larger type to do a load / store of a vector with.
43 unsigned StoreSize = VT.getStoreSizeInBits();
44 if (StoreSize <= 32)
45 return EVT::getIntegerVT(Ctx, StoreSize);
46
47 if (StoreSize % 32 == 0)
48 return EVT::getVectorVT(Ctx, MVT::i32, StoreSize / 32);
49
50 return VT;
51}
52
56
58 // In order for this to be a signed 24-bit value, bit 23, must
59 // be a sign bit.
60 return DAG.ComputeMaxSignificantBits(Op);
61}
62
64 const TargetSubtargetInfo &STI,
65 const AMDGPUSubtarget &AMDGPUSTI)
66 : TargetLowering(TM, STI), Subtarget(&AMDGPUSTI) {
67 // Always lower memset, memcpy, and memmove intrinsics to load/store
68 // instructions, rather then generating calls to memset, mempcy or memmove.
72
73 // Enable ganging up loads and stores in the memcpy DAG lowering.
75
76 // Lower floating point store/load to integer store/load to reduce the number
77 // of patterns in tablegen.
79 AddPromotedToType(ISD::LOAD, MVT::f32, MVT::i32);
80
82 AddPromotedToType(ISD::LOAD, MVT::v2f32, MVT::v2i32);
83
85 AddPromotedToType(ISD::LOAD, MVT::v3f32, MVT::v3i32);
86
88 AddPromotedToType(ISD::LOAD, MVT::v4f32, MVT::v4i32);
89
91 AddPromotedToType(ISD::LOAD, MVT::v5f32, MVT::v5i32);
92
94 AddPromotedToType(ISD::LOAD, MVT::v6f32, MVT::v6i32);
95
97 AddPromotedToType(ISD::LOAD, MVT::v7f32, MVT::v7i32);
98
100 AddPromotedToType(ISD::LOAD, MVT::v8f32, MVT::v8i32);
101
103 AddPromotedToType(ISD::LOAD, MVT::v9f32, MVT::v9i32);
104
105 setOperationAction(ISD::LOAD, MVT::v10f32, Promote);
106 AddPromotedToType(ISD::LOAD, MVT::v10f32, MVT::v10i32);
107
108 setOperationAction(ISD::LOAD, MVT::v11f32, Promote);
109 AddPromotedToType(ISD::LOAD, MVT::v11f32, MVT::v11i32);
110
111 setOperationAction(ISD::LOAD, MVT::v12f32, Promote);
112 AddPromotedToType(ISD::LOAD, MVT::v12f32, MVT::v12i32);
113
114 setOperationAction(ISD::LOAD, MVT::v16f32, Promote);
115 AddPromotedToType(ISD::LOAD, MVT::v16f32, MVT::v16i32);
116
117 setOperationAction(ISD::LOAD, MVT::v32f32, Promote);
118 AddPromotedToType(ISD::LOAD, MVT::v32f32, MVT::v32i32);
119
121 AddPromotedToType(ISD::LOAD, MVT::i64, MVT::v2i32);
122
124 AddPromotedToType(ISD::LOAD, MVT::v2i64, MVT::v4i32);
125
127 AddPromotedToType(ISD::LOAD, MVT::f64, MVT::v2i32);
128
130 AddPromotedToType(ISD::LOAD, MVT::v2f64, MVT::v4i32);
131
133 AddPromotedToType(ISD::LOAD, MVT::v3i64, MVT::v6i32);
134
136 AddPromotedToType(ISD::LOAD, MVT::v4i64, MVT::v8i32);
137
139 AddPromotedToType(ISD::LOAD, MVT::v3f64, MVT::v6i32);
140
142 AddPromotedToType(ISD::LOAD, MVT::v4f64, MVT::v8i32);
143
145 AddPromotedToType(ISD::LOAD, MVT::v8i64, MVT::v16i32);
146
148 AddPromotedToType(ISD::LOAD, MVT::v8f64, MVT::v16i32);
149
150 setOperationAction(ISD::LOAD, MVT::v16i64, Promote);
151 AddPromotedToType(ISD::LOAD, MVT::v16i64, MVT::v32i32);
152
153 setOperationAction(ISD::LOAD, MVT::v16f64, Promote);
154 AddPromotedToType(ISD::LOAD, MVT::v16f64, MVT::v32i32);
155
157 AddPromotedToType(ISD::LOAD, MVT::i128, MVT::v4i32);
158
159 // TODO: Would be better to consume as directly legal
161 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f32, MVT::i32);
162
164 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f64, MVT::i64);
165
167 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::f16, MVT::i16);
168
170 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::bf16, MVT::i16);
171
173 AddPromotedToType(ISD::ATOMIC_LOAD, MVT::v2f32, MVT::i64);
174
176 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f32, MVT::i32);
177
179 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f64, MVT::i64);
180
182 AddPromotedToType(ISD::ATOMIC_STORE, MVT::f16, MVT::i16);
183
185 AddPromotedToType(ISD::ATOMIC_STORE, MVT::bf16, MVT::i16);
186
188 AddPromotedToType(ISD::ATOMIC_STORE, MVT::v2f32, MVT::i64);
189
190 // There are no 64-bit extloads. These should be done as a 32-bit extload and
191 // an extension to 64-bit.
192 for (MVT VT : MVT::integer_valuetypes())
194 Expand);
195
196 for (MVT VT : MVT::integer_valuetypes()) {
197 if (VT == MVT::i64)
198 continue;
199
200 for (auto Op : {ISD::SEXTLOAD, ISD::ZEXTLOAD, ISD::EXTLOAD}) {
201 setLoadExtAction(Op, VT, MVT::i1, Promote);
202 setLoadExtAction(Op, VT, MVT::i8, Legal);
203 setLoadExtAction(Op, VT, MVT::i16, Legal);
204 setLoadExtAction(Op, VT, MVT::i32, Expand);
205 }
206 }
207
209 for (auto MemVT :
210 {MVT::v2i8, MVT::v4i8, MVT::v2i16, MVT::v3i16, MVT::v4i16})
212 Expand);
213
214 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::f16, Expand);
215 setLoadExtAction(ISD::EXTLOAD, MVT::f32, MVT::bf16, Expand);
216 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2f16, Expand);
217 setLoadExtAction(ISD::EXTLOAD, MVT::v2f32, MVT::v2bf16, Expand);
218 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3f16, Expand);
219 setLoadExtAction(ISD::EXTLOAD, MVT::v3f32, MVT::v3bf16, Expand);
220 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4f16, Expand);
221 setLoadExtAction(ISD::EXTLOAD, MVT::v4f32, MVT::v4bf16, Expand);
222 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8f16, Expand);
223 setLoadExtAction(ISD::EXTLOAD, MVT::v8f32, MVT::v8bf16, Expand);
224 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16f16, Expand);
225 setLoadExtAction(ISD::EXTLOAD, MVT::v16f32, MVT::v16bf16, Expand);
226 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32f16, Expand);
227 setLoadExtAction(ISD::EXTLOAD, MVT::v32f32, MVT::v32bf16, Expand);
228
229 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f32, Expand);
230 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f32, Expand);
231 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f32, Expand);
232 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f32, Expand);
233 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f32, Expand);
234 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f32, Expand);
235
236 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::f16, Expand);
237 setLoadExtAction(ISD::EXTLOAD, MVT::f64, MVT::bf16, Expand);
238 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2f16, Expand);
239 setLoadExtAction(ISD::EXTLOAD, MVT::v2f64, MVT::v2bf16, Expand);
240 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3f16, Expand);
241 setLoadExtAction(ISD::EXTLOAD, MVT::v3f64, MVT::v3bf16, Expand);
242 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4f16, Expand);
243 setLoadExtAction(ISD::EXTLOAD, MVT::v4f64, MVT::v4bf16, Expand);
244 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8f16, Expand);
245 setLoadExtAction(ISD::EXTLOAD, MVT::v8f64, MVT::v8bf16, Expand);
246 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16f16, Expand);
247 setLoadExtAction(ISD::EXTLOAD, MVT::v16f64, MVT::v16bf16, Expand);
248
250 AddPromotedToType(ISD::STORE, MVT::f32, MVT::i32);
251
253 AddPromotedToType(ISD::STORE, MVT::v2f32, MVT::v2i32);
254
256 AddPromotedToType(ISD::STORE, MVT::v3f32, MVT::v3i32);
257
259 AddPromotedToType(ISD::STORE, MVT::v4f32, MVT::v4i32);
260
262 AddPromotedToType(ISD::STORE, MVT::v5f32, MVT::v5i32);
263
265 AddPromotedToType(ISD::STORE, MVT::v6f32, MVT::v6i32);
266
268 AddPromotedToType(ISD::STORE, MVT::v7f32, MVT::v7i32);
269
271 AddPromotedToType(ISD::STORE, MVT::v8f32, MVT::v8i32);
272
274 AddPromotedToType(ISD::STORE, MVT::v9f32, MVT::v9i32);
275
277 AddPromotedToType(ISD::STORE, MVT::v10f32, MVT::v10i32);
278
280 AddPromotedToType(ISD::STORE, MVT::v11f32, MVT::v11i32);
281
283 AddPromotedToType(ISD::STORE, MVT::v12f32, MVT::v12i32);
284
286 AddPromotedToType(ISD::STORE, MVT::v16f32, MVT::v16i32);
287
289 AddPromotedToType(ISD::STORE, MVT::v32f32, MVT::v32i32);
290
292 AddPromotedToType(ISD::STORE, MVT::i64, MVT::v2i32);
293
295 AddPromotedToType(ISD::STORE, MVT::v2i64, MVT::v4i32);
296
298 AddPromotedToType(ISD::STORE, MVT::f64, MVT::v2i32);
299
301 AddPromotedToType(ISD::STORE, MVT::v2f64, MVT::v4i32);
302
304 AddPromotedToType(ISD::STORE, MVT::v3i64, MVT::v6i32);
305
307 AddPromotedToType(ISD::STORE, MVT::v3f64, MVT::v6i32);
308
310 AddPromotedToType(ISD::STORE, MVT::v4i64, MVT::v8i32);
311
313 AddPromotedToType(ISD::STORE, MVT::v4f64, MVT::v8i32);
314
316 AddPromotedToType(ISD::STORE, MVT::v8i64, MVT::v16i32);
317
319 AddPromotedToType(ISD::STORE, MVT::v8f64, MVT::v16i32);
320
322 AddPromotedToType(ISD::STORE, MVT::v16i64, MVT::v32i32);
323
325 AddPromotedToType(ISD::STORE, MVT::v16f64, MVT::v32i32);
326
328 AddPromotedToType(ISD::STORE, MVT::i128, MVT::v4i32);
329
330 setTruncStoreAction(MVT::i64, MVT::i1, Expand);
331 setTruncStoreAction(MVT::i64, MVT::i8, Expand);
332 setTruncStoreAction(MVT::i64, MVT::i16, Expand);
333 setTruncStoreAction(MVT::i64, MVT::i32, Expand);
334
335 setTruncStoreAction(MVT::v2i64, MVT::v2i1, Expand);
336 setTruncStoreAction(MVT::v2i64, MVT::v2i8, Expand);
337 setTruncStoreAction(MVT::v2i64, MVT::v2i16, Expand);
338 setTruncStoreAction(MVT::v2i64, MVT::v2i32, Expand);
339
340 setTruncStoreAction(MVT::f32, MVT::bf16, Expand);
341 setTruncStoreAction(MVT::f32, MVT::f16, Expand);
342 setTruncStoreAction(MVT::v2f32, MVT::v2bf16, Expand);
343 setTruncStoreAction(MVT::v2f32, MVT::v2f16, Expand);
344 setTruncStoreAction(MVT::v3f32, MVT::v3bf16, Expand);
345 setTruncStoreAction(MVT::v3f32, MVT::v3f16, Expand);
346 setTruncStoreAction(MVT::v4f32, MVT::v4bf16, Expand);
347 setTruncStoreAction(MVT::v4f32, MVT::v4f16, Expand);
348 setTruncStoreAction(MVT::v6f32, MVT::v6f16, Expand);
349 setTruncStoreAction(MVT::v8f32, MVT::v8bf16, Expand);
350 setTruncStoreAction(MVT::v8f32, MVT::v8f16, Expand);
351 setTruncStoreAction(MVT::v16f32, MVT::v16bf16, Expand);
352 setTruncStoreAction(MVT::v16f32, MVT::v16f16, Expand);
353 setTruncStoreAction(MVT::v32f32, MVT::v32bf16, Expand);
354 setTruncStoreAction(MVT::v32f32, MVT::v32f16, Expand);
355
356 setTruncStoreAction(MVT::f64, MVT::bf16, Expand);
357 setTruncStoreAction(MVT::f64, MVT::f16, Expand);
358 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
359
360 setTruncStoreAction(MVT::v2f64, MVT::v2f32, Expand);
361 setTruncStoreAction(MVT::v2f64, MVT::v2bf16, Expand);
362 setTruncStoreAction(MVT::v2f64, MVT::v2f16, Expand);
363
364 setTruncStoreAction(MVT::v3i32, MVT::v3i8, Expand);
365
366 setTruncStoreAction(MVT::v3i64, MVT::v3i32, Expand);
367 setTruncStoreAction(MVT::v3i64, MVT::v3i16, Expand);
368 setTruncStoreAction(MVT::v3i64, MVT::v3i8, Expand);
369 setTruncStoreAction(MVT::v3i64, MVT::v3i1, Expand);
370 setTruncStoreAction(MVT::v3f64, MVT::v3f32, Expand);
371 setTruncStoreAction(MVT::v3f64, MVT::v3bf16, Expand);
372 setTruncStoreAction(MVT::v3f64, MVT::v3f16, Expand);
373
374 setTruncStoreAction(MVT::v4i64, MVT::v4i32, Expand);
375 setTruncStoreAction(MVT::v4i64, MVT::v4i16, Expand);
376 setTruncStoreAction(MVT::v4f64, MVT::v4f32, Expand);
377 setTruncStoreAction(MVT::v4f64, MVT::v4bf16, Expand);
378 setTruncStoreAction(MVT::v4f64, MVT::v4f16, Expand);
379
380 setTruncStoreAction(MVT::v5i32, MVT::v5i1, Expand);
381 setTruncStoreAction(MVT::v5i32, MVT::v5i8, Expand);
382 setTruncStoreAction(MVT::v5i32, MVT::v5i16, Expand);
383
384 setTruncStoreAction(MVT::v6i32, MVT::v6i1, Expand);
385 setTruncStoreAction(MVT::v6i32, MVT::v6i8, Expand);
386 setTruncStoreAction(MVT::v6i32, MVT::v6i16, Expand);
387
388 setTruncStoreAction(MVT::v7i32, MVT::v7i1, Expand);
389 setTruncStoreAction(MVT::v7i32, MVT::v7i8, Expand);
390 setTruncStoreAction(MVT::v7i32, MVT::v7i16, Expand);
391
392 setTruncStoreAction(MVT::v8f64, MVT::v8f32, Expand);
393 setTruncStoreAction(MVT::v8f64, MVT::v8bf16, Expand);
394 setTruncStoreAction(MVT::v8f64, MVT::v8f16, Expand);
395
396 setTruncStoreAction(MVT::v16f64, MVT::v16f32, Expand);
397 setTruncStoreAction(MVT::v16f64, MVT::v16bf16, Expand);
398 setTruncStoreAction(MVT::v16f64, MVT::v16f16, Expand);
399 setTruncStoreAction(MVT::v16i64, MVT::v16i16, Expand);
400 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
401 setTruncStoreAction(MVT::v16i64, MVT::v16i8, Expand);
402 setTruncStoreAction(MVT::v16i64, MVT::v16i1, Expand);
403
404 setOperationAction(ISD::Constant, {MVT::i32, MVT::i64}, Legal);
405 setOperationAction(ISD::ConstantFP, {MVT::f32, MVT::f64}, Legal);
406
408
409 // For R600, this is totally unsupported, just custom lower to produce an
410 // error.
412
413 // Library functions. These default to Expand, but we have instructions
414 // for them.
417 {MVT::f16, MVT::f32}, Legal);
419
421 setOperationAction(ISD::FROUND, {MVT::f32, MVT::f64}, Custom);
423 {MVT::f16, MVT::f32, MVT::f64}, Expand);
424
427 Custom);
429
430 setOperationAction(ISD::FNEARBYINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
431
432 setOperationAction(ISD::FRINT, {MVT::f16, MVT::f32, MVT::f64}, Custom);
433
434 setOperationAction({ISD::LRINT, ISD::LLRINT}, {MVT::f16, MVT::f32, MVT::f64},
435 Expand);
436
437 setOperationAction(ISD::FREM, {MVT::f16, MVT::f32, MVT::f64}, Expand);
438 setOperationAction(ISD::IS_FPCLASS, {MVT::f32, MVT::f64}, Legal);
440
442 Custom);
443
444 setOperationAction(ISD::FCANONICALIZE, {MVT::f32, MVT::f64}, Legal);
445
446 // FIXME: These IS_FPCLASS vector fp types are marked custom so it reaches
447 // scalarization code. Can be removed when IS_FPCLASS expand isn't called by
448 // default unless marked custom/legal.
450 {MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
451 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v16f32,
452 MVT::v2f64, MVT::v3f64, MVT::v4f64, MVT::v8f64,
453 MVT::v16f64},
454 Custom);
455
456 // Expand to fneg + fadd.
458
460 {MVT::v3i32, MVT::v3f32, MVT::v4i32, MVT::v4f32,
461 MVT::v5i32, MVT::v5f32, MVT::v6i32, MVT::v6f32,
462 MVT::v7i32, MVT::v7f32, MVT::v8i32, MVT::v8f32,
463 MVT::v9i32, MVT::v9f32, MVT::v10i32, MVT::v10f32,
464 MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32},
465 Custom);
466
469 {MVT::v2f32, MVT::v2i32, MVT::v3f32, MVT::v3i32, MVT::v4f32,
470 MVT::v4i32, MVT::v5f32, MVT::v5i32, MVT::v6f32, MVT::v6i32,
471 MVT::v7f32, MVT::v7i32, MVT::v8f32, MVT::v8i32, MVT::v9f32,
472 MVT::v9i32, MVT::v10i32, MVT::v10f32, MVT::v11i32, MVT::v11f32,
473 MVT::v12i32, MVT::v12f32, MVT::v16i32, MVT::v32f32, MVT::v32i32,
474 MVT::v2f64, MVT::v2i64, MVT::v3f64, MVT::v3i64, MVT::v4f64,
475 MVT::v4i64, MVT::v8f64, MVT::v8i64, MVT::v16f64, MVT::v16i64},
476 Custom);
477
479 Expand);
480 setOperationAction(ISD::FP_TO_FP16, {MVT::f64, MVT::f32}, Custom);
481
482 const MVT ScalarIntVTs[] = { MVT::i32, MVT::i64 };
483 for (MVT VT : ScalarIntVTs) {
484 // These should use [SU]DIVREM, so set them to expand
486 Expand);
487
488 // GPU does not have divrem function for signed or unsigned.
490
491 // GPU does not have [S|U]MUL_LOHI functions as a single instruction.
493
495
497 Expand);
498 }
499
500 // The hardware supports 32-bit FSHR, but not FSHL.
502
503 setOperationAction({ISD::ROTL, ISD::ROTR}, {MVT::i32, MVT::i64}, Expand);
504
506
511 MVT::i64, Custom);
513
515 Legal);
516
519 MVT::i64, Custom);
520
521 for (auto VT : {MVT::i8, MVT::i16})
523
524 static const MVT::SimpleValueType VectorIntTypes[] = {
525 MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32, MVT::v6i32, MVT::v7i32,
526 MVT::v9i32, MVT::v10i32, MVT::v11i32, MVT::v12i32};
527
528 for (MVT VT : VectorIntTypes) {
529 // Expand the following operations for the current type by default.
530 // clang-format off
550 VT, Expand);
551 // clang-format on
552 }
553
554 static const MVT::SimpleValueType FloatVectorTypes[] = {
555 MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32, MVT::v6f32, MVT::v7f32,
556 MVT::v9f32, MVT::v10f32, MVT::v11f32, MVT::v12f32};
557
558 for (MVT VT : FloatVectorTypes) {
571 VT, Expand);
572 }
573
574 // This causes using an unrolled select operation rather than expansion with
575 // bit operations. This is in general better, but the alternative using BFI
576 // instructions may be better if the select sources are SGPRs.
578 AddPromotedToType(ISD::SELECT, MVT::v2f32, MVT::v2i32);
579
581 AddPromotedToType(ISD::SELECT, MVT::v3f32, MVT::v3i32);
582
584 AddPromotedToType(ISD::SELECT, MVT::v4f32, MVT::v4i32);
585
587 AddPromotedToType(ISD::SELECT, MVT::v5f32, MVT::v5i32);
588
590 AddPromotedToType(ISD::SELECT, MVT::v6f32, MVT::v6i32);
591
593 AddPromotedToType(ISD::SELECT, MVT::v7f32, MVT::v7i32);
594
596 AddPromotedToType(ISD::SELECT, MVT::v9f32, MVT::v9i32);
597
599 AddPromotedToType(ISD::SELECT, MVT::v10f32, MVT::v10i32);
600
602 AddPromotedToType(ISD::SELECT, MVT::v11f32, MVT::v11i32);
603
605 AddPromotedToType(ISD::SELECT, MVT::v12f32, MVT::v12i32);
606
608 setJumpIsExpensive(true);
609
612
614
615 // We want to find all load dependencies for long chains of stores to enable
616 // merging into very wide vectors. The problem is with vectors with > 4
617 // elements. MergeConsecutiveStores will attempt to merge these because x8/x16
618 // vectors are a legal type, even though we have to split the loads
619 // usually. When we can more precisely specify load legality per address
620 // space, we should be able to make FindBetterChain/MergeConsecutiveStores
621 // smarter so that they can figure out what to do in 2 iterations without all
622 // N > 4 stores on the same chain.
624
625 // memcpy/memmove/memset are expanded in the IR, so we shouldn't need to worry
626 // about these during lowering.
627 MaxStoresPerMemcpy = 0xffffffff;
628 MaxStoresPerMemmove = 0xffffffff;
629 MaxStoresPerMemset = 0xffffffff;
630
631 // The expansion for 64-bit division is enormous.
633 addBypassSlowDiv(64, 32);
634
645
649}
650
651//===----------------------------------------------------------------------===//
652// Target Information
653//===----------------------------------------------------------------------===//
654
656static bool fnegFoldsIntoOpcode(unsigned Opc) {
657 switch (Opc) {
658 case ISD::FADD:
659 case ISD::FSUB:
660 case ISD::FMUL:
661 case ISD::FMA:
662 case ISD::FMAD:
663 case ISD::FMINNUM:
664 case ISD::FMAXNUM:
667 case ISD::FMINIMUM:
668 case ISD::FMAXIMUM:
669 case ISD::FMINIMUMNUM:
670 case ISD::FMAXIMUMNUM:
671 case ISD::SELECT:
672 case ISD::FSIN:
673 case ISD::FTRUNC:
674 case ISD::FRINT:
675 case ISD::FNEARBYINT:
676 case ISD::FROUNDEVEN:
678 case AMDGPUISD::RCP:
679 case AMDGPUISD::RCP_LEGACY:
680 case AMDGPUISD::RCP_IFLAG:
681 case AMDGPUISD::SIN_HW:
682 case AMDGPUISD::FMUL_LEGACY:
683 case AMDGPUISD::FMIN_LEGACY:
684 case AMDGPUISD::FMAX_LEGACY:
685 case AMDGPUISD::FMED3:
686 // TODO: handle llvm.amdgcn.fma.legacy
687 return true;
688 case ISD::BITCAST:
689 llvm_unreachable("bitcast is special cased");
690 default:
691 return false;
692 }
693}
694
695static bool fnegFoldsIntoOp(const SDNode *N) {
696 unsigned Opc = N->getOpcode();
697 if (Opc == ISD::BITCAST) {
698 // TODO: Is there a benefit to checking the conditions performFNegCombine
699 // does? We don't for the other cases.
700 SDValue BCSrc = N->getOperand(0);
701 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
702 return BCSrc.getNumOperands() == 2 &&
703 BCSrc.getOperand(1).getValueSizeInBits() == 32;
704 }
705
706 return BCSrc.getOpcode() == ISD::SELECT && BCSrc.getValueType() == MVT::f32;
707 }
708
709 return fnegFoldsIntoOpcode(Opc);
710}
711
712/// \p returns true if the operation will definitely need to use a 64-bit
713/// encoding, and thus will use a VOP3 encoding regardless of the source
714/// modifiers.
716static bool opMustUseVOP3Encoding(const SDNode *N, MVT VT) {
717 return (N->getNumOperands() > 2 && N->getOpcode() != ISD::SELECT) ||
718 VT == MVT::f64;
719}
720
721/// Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the
722/// type for ISD::SELECT.
724static bool selectSupportsSourceMods(const SDNode *N) {
725 // TODO: Only applies if select will be vector
726 return N->getValueType(0) == MVT::f32;
727}
728
729// Most FP instructions support source modifiers, but this could be refined
730// slightly.
732static bool hasSourceMods(const SDNode *N) {
733 if (isa<MemSDNode>(N))
734 return false;
735
736 switch (N->getOpcode()) {
737 case ISD::CopyToReg:
738 case ISD::FDIV:
739 case ISD::FREM:
740 case ISD::INLINEASM:
742 case AMDGPUISD::DIV_SCALE:
744
745 // TODO: Should really be looking at the users of the bitcast. These are
746 // problematic because bitcasts are used to legalize all stores to integer
747 // types.
748 case ISD::BITCAST:
749 return false;
751 switch (N->getConstantOperandVal(0)) {
752 case Intrinsic::amdgcn_interp_p1:
753 case Intrinsic::amdgcn_interp_p2:
754 case Intrinsic::amdgcn_interp_mov:
755 case Intrinsic::amdgcn_interp_p1_f16:
756 case Intrinsic::amdgcn_interp_p2_f16:
757 return false;
758 default:
759 return true;
760 }
761 }
762 case ISD::SELECT:
764 default:
765 return true;
766 }
767}
768
770 unsigned CostThreshold) {
771 // Some users (such as 3-operand FMA/MAD) must use a VOP3 encoding, and thus
772 // it is truly free to use a source modifier in all cases. If there are
773 // multiple users but for each one will necessitate using VOP3, there will be
774 // a code size increase. Try to avoid increasing code size unless we know it
775 // will save on the instruction count.
776 unsigned NumMayIncreaseSize = 0;
777 MVT VT = N->getValueType(0).getScalarType().getSimpleVT();
778
779 assert(!N->use_empty());
780
781 // XXX - Should this limit number of uses to check?
782 for (const SDNode *U : N->users()) {
783 if (!hasSourceMods(U))
784 return false;
785
786 if (!opMustUseVOP3Encoding(U, VT)) {
787 if (++NumMayIncreaseSize > CostThreshold)
788 return false;
789 }
790 }
791
792 return true;
793}
794
796 ISD::NodeType ExtendKind) const {
797 assert(!VT.isVector() && "only scalar expected");
798
799 // Round to the next multiple of 32-bits.
800 unsigned Size = VT.getSizeInBits();
801 if (Size <= 32)
802 return MVT::i32;
803 return EVT::getIntegerVT(Context, 32 * ((Size + 31) / 32));
804}
805
807 return 32;
808}
809
811 return true;
812}
813
814// The backend supports 32 and 64 bit floating point immediates.
815// FIXME: Why are we reporting vectors of FP immediates as legal?
817 bool ForCodeSize) const {
818 return isTypeLegal(VT.getScalarType());
819}
820
821// We don't want to shrink f64 / f32 constants.
823 EVT ScalarVT = VT.getScalarType();
824 return (ScalarVT != MVT::f32 && ScalarVT != MVT::f64);
825}
826
828 SDNode *N, ISD::LoadExtType ExtTy, EVT NewVT,
829 std::optional<unsigned> ByteOffset) const {
830 // TODO: This may be worth removing. Check regression tests for diffs.
831 if (!TargetLoweringBase::shouldReduceLoadWidth(N, ExtTy, NewVT, ByteOffset))
832 return false;
833
834 unsigned NewSize = NewVT.getStoreSizeInBits();
835
836 // If we are reducing to a 32-bit load or a smaller multi-dword load,
837 // this is always better.
838 if (NewSize >= 32)
839 return true;
840
841 EVT OldVT = N->getValueType(0);
842 unsigned OldSize = OldVT.getStoreSizeInBits();
843
845 unsigned AS = MN->getAddressSpace();
846 // Do not shrink an aligned scalar load to sub-dword.
847 // Scalar engine cannot do sub-dword loads.
848 // TODO: Update this for GFX12 which does have scalar sub-dword loads.
849 if (OldSize >= 32 && NewSize < 32 && MN->getAlign() >= Align(4) &&
853 MN->isInvariant())) &&
855 return false;
856
857 // Don't produce extloads from sub 32-bit types. SI doesn't have scalar
858 // extloads, so doing one requires using a buffer_load. In cases where we
859 // still couldn't use a scalar load, using the wider load shouldn't really
860 // hurt anything.
861
862 // If the old size already had to be an extload, there's no harm in continuing
863 // to reduce the width.
864 return (OldSize < 32);
865}
866
868 const SelectionDAG &DAG,
869 const MachineMemOperand &MMO) const {
870
871 assert(LoadTy.getSizeInBits() == CastTy.getSizeInBits());
872
873 if (LoadTy.getScalarType() == MVT::i32)
874 return false;
875
876 unsigned LScalarSize = LoadTy.getScalarSizeInBits();
877 unsigned CastScalarSize = CastTy.getScalarSizeInBits();
878
879 if ((LScalarSize >= CastScalarSize) && (CastScalarSize < 32))
880 return false;
881
882 unsigned Fast = 0;
884 CastTy, MMO, &Fast) &&
885 Fast;
886}
887
888// SI+ has instructions for cttz / ctlz for 32-bit values. This is probably also
889// profitable with the expansion for 64-bit since it's generally good to
890// speculate things.
892 return true;
893}
894
896 return true;
897}
898
900 switch (N->getOpcode()) {
901 case ISD::EntryToken:
902 case ISD::TokenFactor:
903 return true;
905 unsigned IntrID = N->getConstantOperandVal(0);
907 }
909 unsigned IntrID = N->getConstantOperandVal(1);
911 }
912 case ISD::LOAD:
913 if (cast<LoadSDNode>(N)->getMemOperand()->getAddrSpace() ==
915 return true;
916 return false;
917 case AMDGPUISD::SETCC: // ballot-style instruction
918 return true;
919 }
920 return false;
921}
922
924 SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize,
925 NegatibleCost &Cost, unsigned Depth) const {
926
927 switch (Op.getOpcode()) {
928 case ISD::FMA:
929 case ISD::FMAD: {
930 // Negating a fma is not free if it has users without source mods.
931 if (!allUsesHaveSourceMods(Op.getNode()))
932 return SDValue();
933 break;
934 }
935 case AMDGPUISD::RCP: {
936 SDValue Src = Op.getOperand(0);
937 EVT VT = Op.getValueType();
938 SDLoc SL(Op);
939
940 SDValue NegSrc = getNegatedExpression(Src, DAG, LegalOperations,
941 ForCodeSize, Cost, Depth + 1);
942 if (NegSrc)
943 return DAG.getNode(AMDGPUISD::RCP, SL, VT, NegSrc, Op->getFlags());
944 return SDValue();
945 }
946 default:
947 break;
948 }
949
950 return TargetLowering::getNegatedExpression(Op, DAG, LegalOperations,
951 ForCodeSize, Cost, Depth);
952}
953
954//===---------------------------------------------------------------------===//
955// Target Properties
956//===---------------------------------------------------------------------===//
957
960
961 // Packed operations do not have a fabs modifier.
962 // Report this based on the end legalized type.
963 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
964}
965
968 // Report this based on the end legalized type.
969 VT = VT.getScalarType();
970 return VT == MVT::f32 || VT == MVT::f64 || VT == MVT::f16 || VT == MVT::bf16;
971}
972
974 unsigned NumElem,
975 unsigned AS) const {
976 return true;
977}
978
980 // There are few operations which truly have vector input operands. Any vector
981 // operation is going to involve operations on each component, and a
982 // build_vector will be a copy per element, so it always makes sense to use a
983 // build_vector input in place of the extracted element to avoid a copy into a
984 // super register.
985 //
986 // We should probably only do this if all users are extracts only, but this
987 // should be the common case.
988 return true;
989}
990
992 // Truncate is just accessing a subregister.
993
994 unsigned SrcSize = Source.getSizeInBits();
995 unsigned DestSize = Dest.getSizeInBits();
996
997 return DestSize < SrcSize && DestSize % 32 == 0 ;
998}
999
1001 // Truncate is just accessing a subregister.
1002
1003 unsigned SrcSize = Source->getScalarSizeInBits();
1004 unsigned DestSize = Dest->getScalarSizeInBits();
1005
1006 if (DestSize== 16 && Subtarget->has16BitInsts())
1007 return SrcSize >= 32;
1008
1009 return DestSize < SrcSize && DestSize % 32 == 0;
1010}
1011
1013 unsigned SrcSize = Src->getScalarSizeInBits();
1014 unsigned DestSize = Dest->getScalarSizeInBits();
1015
1016 if (SrcSize == 16 && Subtarget->has16BitInsts())
1017 return DestSize >= 32;
1018
1019 return SrcSize == 32 && DestSize == 64;
1020}
1021
1023 // Any register load of a 64-bit value really requires 2 32-bit moves. For all
1024 // practical purposes, the extra mov 0 to load a 64-bit is free. As used,
1025 // this will enable reducing 64-bit operations the 32-bit, which is always
1026 // good.
1027
1028 if (Src == MVT::i16)
1029 return Dest == MVT::i32 ||Dest == MVT::i64 ;
1030
1031 return Src == MVT::i32 && Dest == MVT::i64;
1032}
1033
1035 EVT DestVT) const {
1036 switch (N->getOpcode()) {
1037 case ISD::ABS:
1038 case ISD::ADD:
1039 case ISD::SUB:
1040 case ISD::SHL:
1041 case ISD::SRL:
1042 case ISD::SRA:
1043 case ISD::AND:
1044 case ISD::OR:
1045 case ISD::XOR:
1046 case ISD::MUL:
1047 case ISD::SETCC:
1048 case ISD::SELECT:
1049 case ISD::SMIN:
1050 case ISD::SMAX:
1051 case ISD::UMIN:
1052 case ISD::UMAX:
1053 case ISD::USUBSAT:
1054 case ISD::UADDSAT:
1055 if (isTypeLegal(MVT::i16) &&
1056 (!DestVT.isVector() ||
1057 !isOperationLegal(ISD::ADD, MVT::v2i16))) { // Check if VOP3P
1058 // Don't narrow back down to i16 if promoted to i32 already.
1059 if (!N->isDivergent() && DestVT.isInteger() &&
1060 DestVT.getScalarSizeInBits() > 1 &&
1061 DestVT.getScalarSizeInBits() <= 16 &&
1062 SrcVT.getScalarSizeInBits() > 16) {
1063 return false;
1064 }
1065 }
1066 return true;
1067 default:
1068 break;
1069 }
1070
1071 // There aren't really 64-bit registers, but pairs of 32-bit ones and only a
1072 // limited number of native 64-bit operations. Shrinking an operation to fit
1073 // in a single 32-bit register should always be helpful. As currently used,
1074 // this is much less general than the name suggests, and is only used in
1075 // places trying to reduce the sizes of loads. Shrinking loads to < 32-bits is
1076 // not profitable, and may actually be harmful.
1077 if (isa<LoadSDNode>(N))
1078 return SrcVT.getSizeInBits() > 32 && DestVT.getSizeInBits() == 32;
1079
1080 return true;
1081}
1082
1084 const SDNode* N, CombineLevel Level) const {
1085 assert((N->getOpcode() == ISD::SHL || N->getOpcode() == ISD::SRA ||
1086 N->getOpcode() == ISD::SRL) &&
1087 "Expected shift op");
1088
1089 SDValue ShiftLHS = N->getOperand(0);
1090 if (!ShiftLHS->hasOneUse())
1091 return false;
1092
1093 if (ShiftLHS.getOpcode() == ISD::SIGN_EXTEND &&
1094 !ShiftLHS.getOperand(0)->hasOneUse())
1095 return false;
1096
1097 // Always commute pre-type legalization and right shifts.
1098 // We're looking for shl(or(x,y),z) patterns.
1100 N->getOpcode() != ISD::SHL || N->getOperand(0).getOpcode() != ISD::OR)
1101 return true;
1102
1103 // If only user is a i32 right-shift, then don't destroy a BFE pattern.
1104 if (N->getValueType(0) == MVT::i32 && N->hasOneUse() &&
1105 (N->user_begin()->getOpcode() == ISD::SRA ||
1106 N->user_begin()->getOpcode() == ISD::SRL))
1107 return false;
1108
1109 // Don't destroy or(shl(load_zext(),c), load_zext()) patterns.
1110 auto IsShiftAndLoad = [](SDValue LHS, SDValue RHS) {
1111 if (LHS.getOpcode() != ISD::SHL)
1112 return false;
1113 auto *RHSLd = dyn_cast<LoadSDNode>(RHS);
1114 auto *LHS0 = dyn_cast<LoadSDNode>(LHS.getOperand(0));
1115 auto *LHS1 = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
1116 return LHS0 && LHS1 && RHSLd && LHS0->getExtensionType() == ISD::ZEXTLOAD &&
1117 LHS1->getAPIntValue() == LHS0->getMemoryVT().getScalarSizeInBits() &&
1118 RHSLd->getExtensionType() == ISD::ZEXTLOAD;
1119 };
1120 SDValue LHS = N->getOperand(0).getOperand(0);
1121 SDValue RHS = N->getOperand(0).getOperand(1);
1122 return !(IsShiftAndLoad(LHS, RHS) || IsShiftAndLoad(RHS, LHS));
1123}
1124
1125//===---------------------------------------------------------------------===//
1126// TargetLowering Callbacks
1127//===---------------------------------------------------------------------===//
1128
1130 bool IsVarArg) {
1131 switch (CC) {
1139 return CC_AMDGPU;
1142 return CC_AMDGPU_CS_CHAIN;
1143 case CallingConv::C:
1144 case CallingConv::Fast:
1145 case CallingConv::Cold:
1146 return CC_AMDGPU_Func;
1149 return CC_SI_Gfx;
1152 default:
1153 reportFatalUsageError("unsupported calling convention for call");
1154 }
1155}
1156
1158 bool IsVarArg) {
1159 switch (CC) {
1162 llvm_unreachable("kernels should not be handled here");
1172 return RetCC_SI_Shader;
1175 return RetCC_SI_Gfx;
1176 case CallingConv::C:
1177 case CallingConv::Fast:
1178 case CallingConv::Cold:
1179 return RetCC_AMDGPU_Func;
1180 default:
1181 reportFatalUsageError("unsupported calling convention");
1182 }
1183}
1184
1185/// The SelectionDAGBuilder will automatically promote function arguments
1186/// with illegal types. However, this does not work for the AMDGPU targets
1187/// since the function arguments are stored in memory as these illegal types.
1188/// In order to handle this properly we need to get the original types sizes
1189/// from the LLVM IR Function and fixup the ISD:InputArg values before
1190/// passing them to AnalyzeFormalArguments()
1191
1192/// When the SelectionDAGBuilder computes the Ins, it takes care of splitting
1193/// input values across multiple registers. Each item in the Ins array
1194/// represents a single value that will be stored in registers. Ins[x].VT is
1195/// the value type of the value that will be stored in the register, so
1196/// whatever SDNode we lower the argument to needs to be this type.
1197///
1198/// In order to correctly lower the arguments we need to know the size of each
1199/// argument. Since Ins[x].VT gives us the size of the register that will
1200/// hold the value, we need to look at Ins[x].ArgVT to see the 'real' type
1201/// for the original function argument so that we can deduce the correct memory
1202/// type to use for Ins[x]. In most cases the correct memory type will be
1203/// Ins[x].ArgVT. However, this will not always be the case. If, for example,
1204/// we have a kernel argument of type v8i8, this argument will be split into
1205/// 8 parts and each part will be represented by its own item in the Ins array.
1206/// For each part the Ins[x].ArgVT will be the v8i8, which is the full type of
1207/// the argument before it was split. From this, we deduce that the memory type
1208/// for each individual part is i8. We pass the memory type as LocVT to the
1209/// calling convention analysis function and the register type (Ins[x].VT) as
1210/// the ValVT.
1212 CCState &State,
1213 const SmallVectorImpl<ISD::InputArg> &Ins) const {
1214 const MachineFunction &MF = State.getMachineFunction();
1215 const Function &Fn = MF.getFunction();
1216 LLVMContext &Ctx = Fn.getContext();
1217 const unsigned ExplicitOffset = Subtarget->getExplicitKernelArgOffset();
1219
1220 Align MaxAlign = Align(1);
1221 uint64_t ExplicitArgOffset = 0;
1222 const DataLayout &DL = Fn.getDataLayout();
1223
1224 unsigned InIndex = 0;
1225
1226 for (const Argument &Arg : Fn.args()) {
1227 const bool IsByRef = Arg.hasByRefAttr();
1228 Type *BaseArgTy = Arg.getType();
1229 Type *MemArgTy = IsByRef ? Arg.getParamByRefType() : BaseArgTy;
1230 Align Alignment = DL.getValueOrABITypeAlignment(
1231 IsByRef ? Arg.getParamAlign() : std::nullopt, MemArgTy);
1232 MaxAlign = std::max(Alignment, MaxAlign);
1233 uint64_t AllocSize = DL.getTypeAllocSize(MemArgTy);
1234
1235 uint64_t ArgOffset = alignTo(ExplicitArgOffset, Alignment) + ExplicitOffset;
1236 ExplicitArgOffset = alignTo(ExplicitArgOffset, Alignment) + AllocSize;
1237
1238 // We're basically throwing away everything passed into us and starting over
1239 // to get accurate in-memory offsets. The "PartOffset" is completely useless
1240 // to us as computed in Ins.
1241 //
1242 // We also need to figure out what type legalization is trying to do to get
1243 // the correct memory offsets.
1244
1245 SmallVector<EVT, 16> ValueVTs;
1247 ComputeValueVTs(*this, DL, BaseArgTy, ValueVTs, /*MemVTs=*/nullptr,
1248 &Offsets, ArgOffset);
1249
1250 for (unsigned Value = 0, NumValues = ValueVTs.size();
1251 Value != NumValues; ++Value) {
1252 uint64_t BasePartOffset = Offsets[Value];
1253
1254 EVT ArgVT = ValueVTs[Value];
1255 EVT MemVT = ArgVT;
1256 MVT RegisterVT = getRegisterTypeForCallingConv(Ctx, CC, ArgVT);
1257 unsigned NumRegs = getNumRegistersForCallingConv(Ctx, CC, ArgVT);
1258
1259 if (NumRegs == 1) {
1260 // This argument is not split, so the IR type is the memory type.
1261 if (ArgVT.isExtended()) {
1262 // We have an extended type, like i24, so we should just use the
1263 // register type.
1264 MemVT = RegisterVT;
1265 } else {
1266 MemVT = ArgVT;
1267 }
1268 } else if (ArgVT.isVector() && RegisterVT.isVector() &&
1269 ArgVT.getScalarType() == RegisterVT.getScalarType()) {
1270 assert(ArgVT.getVectorNumElements() > RegisterVT.getVectorNumElements());
1271 // We have a vector value which has been split into a vector with
1272 // the same scalar type, but fewer elements. This should handle
1273 // all the floating-point vector types.
1274 MemVT = RegisterVT;
1275 } else if (ArgVT.isVector() &&
1276 ArgVT.getVectorNumElements() == NumRegs) {
1277 // This arg has been split so that each element is stored in a separate
1278 // register.
1279 MemVT = ArgVT.getScalarType();
1280 } else if (ArgVT.isExtended()) {
1281 // We have an extended type, like i65.
1282 MemVT = RegisterVT;
1283 } else {
1284 unsigned MemoryBits = ArgVT.getStoreSizeInBits() / NumRegs;
1285 assert(ArgVT.getStoreSizeInBits() % NumRegs == 0);
1286 if (RegisterVT.isInteger()) {
1287 MemVT = EVT::getIntegerVT(State.getContext(), MemoryBits);
1288 } else if (RegisterVT.isVector()) {
1289 assert(!RegisterVT.getScalarType().isFloatingPoint());
1290 unsigned NumElements = RegisterVT.getVectorNumElements();
1291 assert(MemoryBits % NumElements == 0);
1292 // This vector type has been split into another vector type with
1293 // a different elements size.
1294 EVT ScalarVT = EVT::getIntegerVT(State.getContext(),
1295 MemoryBits / NumElements);
1296 MemVT = EVT::getVectorVT(State.getContext(), ScalarVT, NumElements);
1297 } else {
1298 llvm_unreachable("cannot deduce memory type.");
1299 }
1300 }
1301
1302 // Convert one element vectors to scalar.
1303 if (MemVT.isVector() && MemVT.getVectorNumElements() == 1)
1304 MemVT = MemVT.getScalarType();
1305
1306 // Round up vec3/vec5 argument.
1307 if (MemVT.isVector() && !MemVT.isPow2VectorType()) {
1308 MemVT = MemVT.getPow2VectorType(State.getContext());
1309 } else if (!MemVT.isSimple() && !MemVT.isVector()) {
1310 MemVT = MemVT.getRoundIntegerType(State.getContext());
1311 }
1312
1313 unsigned PartOffset = 0;
1314 for (unsigned i = 0; i != NumRegs; ++i) {
1315 State.addLoc(CCValAssign::getCustomMem(InIndex++, RegisterVT,
1316 BasePartOffset + PartOffset,
1317 MemVT.getSimpleVT(),
1319 PartOffset += MemVT.getStoreSize();
1320 }
1321 }
1322 }
1323}
1324
1326 SDValue Chain, CallingConv::ID CallConv,
1327 bool isVarArg,
1329 const SmallVectorImpl<SDValue> &OutVals,
1330 const SDLoc &DL, SelectionDAG &DAG) const {
1331 // FIXME: Fails for r600 tests
1332 //assert(!isVarArg && Outs.empty() && OutVals.empty() &&
1333 // "wave terminate should not have return values");
1334 return DAG.getNode(AMDGPUISD::ENDPGM, DL, MVT::Other, Chain);
1335}
1336
1337//===---------------------------------------------------------------------===//
1338// Target specific lowering
1339//===---------------------------------------------------------------------===//
1340
1341/// Selects the correct CCAssignFn for a given CallingConvention value.
1346
1351
1353 SelectionDAG &DAG,
1354 MachineFrameInfo &MFI,
1355 int ClobberedFI) const {
1356 SmallVector<SDValue, 8> ArgChains;
1357 int64_t FirstByte = MFI.getObjectOffset(ClobberedFI);
1358 int64_t LastByte = FirstByte + MFI.getObjectSize(ClobberedFI) - 1;
1359
1360 // Include the original chain at the beginning of the list. When this is
1361 // used by target LowerCall hooks, this helps legalize find the
1362 // CALLSEQ_BEGIN node.
1363 ArgChains.push_back(Chain);
1364
1365 // Add a chain value for each stack argument corresponding
1366 for (SDNode *U : DAG.getEntryNode().getNode()->users()) {
1367 if (LoadSDNode *L = dyn_cast<LoadSDNode>(U)) {
1368 if (FrameIndexSDNode *FI = dyn_cast<FrameIndexSDNode>(L->getBasePtr())) {
1369 if (FI->getIndex() < 0) {
1370 int64_t InFirstByte = MFI.getObjectOffset(FI->getIndex());
1371 int64_t InLastByte = InFirstByte;
1372 InLastByte += MFI.getObjectSize(FI->getIndex()) - 1;
1373
1374 if ((InFirstByte <= FirstByte && FirstByte <= InLastByte) ||
1375 (FirstByte <= InFirstByte && InFirstByte <= LastByte))
1376 ArgChains.push_back(SDValue(L, 1));
1377 }
1378 }
1379 }
1380 }
1381
1382 // Build a tokenfactor for all the chains.
1383 return DAG.getNode(ISD::TokenFactor, SDLoc(Chain), MVT::Other, ArgChains);
1384}
1385
1388 StringRef Reason) const {
1389 SDValue Callee = CLI.Callee;
1390 SelectionDAG &DAG = CLI.DAG;
1391
1392 const Function &Fn = DAG.getMachineFunction().getFunction();
1393
1394 StringRef FuncName("<unknown>");
1395
1397 FuncName = G->getSymbol();
1398 else if (const GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee))
1399 FuncName = G->getGlobal()->getName();
1400
1401 DAG.getContext()->diagnose(
1402 DiagnosticInfoUnsupported(Fn, Reason + FuncName, CLI.DL.getDebugLoc()));
1403
1404 if (!CLI.IsTailCall) {
1405 for (ISD::InputArg &Arg : CLI.Ins)
1406 InVals.push_back(DAG.getPOISON(Arg.VT));
1407 }
1408
1409 // FIXME: Hack because R600 doesn't handle callseq pseudos yet.
1410 if (getTargetMachine().getTargetTriple().getArch() == Triple::r600)
1411 return CLI.Chain;
1412
1413 SDValue Chain = DAG.getCALLSEQ_START(CLI.Chain, 0, 0, CLI.DL);
1414 return DAG.getCALLSEQ_END(Chain, 0, 0, /*InGlue=*/SDValue(), CLI.DL);
1415}
1416
1418 SmallVectorImpl<SDValue> &InVals) const {
1419 return lowerUnhandledCall(CLI, InVals, "unsupported call to function ");
1420}
1421
1423 SelectionDAG &DAG) const {
1424 const Function &Fn = DAG.getMachineFunction().getFunction();
1425
1427 Fn, "unsupported dynamic alloca", SDLoc(Op).getDebugLoc()));
1428 auto Ops = {DAG.getConstant(0, SDLoc(), Op.getValueType()), Op.getOperand(0)};
1429 return DAG.getMergeValues(Ops, SDLoc());
1430}
1431
1433 SelectionDAG &DAG) const {
1434 switch (Op.getOpcode()) {
1435 default:
1436 Op->print(errs(), &DAG);
1437 llvm_unreachable("Custom lowering code for this "
1438 "instruction is not implemented yet!");
1439 break;
1441 case ISD::CONCAT_VECTORS: return LowerCONCAT_VECTORS(Op, DAG);
1443 case ISD::UDIVREM: return LowerUDIVREM(Op, DAG);
1444 case ISD::SDIVREM:
1445 return LowerSDIVREM(Op, DAG);
1446 case ISD::FCEIL: return LowerFCEIL(Op, DAG);
1447 case ISD::FTRUNC: return LowerFTRUNC(Op, DAG);
1448 case ISD::FRINT: return LowerFRINT(Op, DAG);
1449 case ISD::FNEARBYINT: return LowerFNEARBYINT(Op, DAG);
1450 case ISD::FROUNDEVEN:
1451 return LowerFROUNDEVEN(Op, DAG);
1452 case ISD::FROUND: return LowerFROUND(Op, DAG);
1453 case ISD::FFLOOR: return LowerFFLOOR(Op, DAG);
1454 case ISD::FLOG2:
1455 return LowerFLOG2(Op, DAG);
1456 case ISD::FLOG:
1457 case ISD::FLOG10:
1458 return LowerFLOGCommon(Op, DAG);
1459 case ISD::FEXP:
1460 case ISD::FEXP10:
1461 return lowerFEXP(Op, DAG);
1462 case ISD::FEXP2:
1463 return lowerFEXP2(Op, DAG);
1464 case ISD::SINT_TO_FP: return LowerSINT_TO_FP(Op, DAG);
1465 case ISD::UINT_TO_FP: return LowerUINT_TO_FP(Op, DAG);
1466 case ISD::FP_TO_FP16: return LowerFP_TO_FP16(Op, DAG);
1467 case ISD::FP_TO_SINT:
1468 case ISD::FP_TO_UINT:
1469 return LowerFP_TO_INT(Op, DAG);
1472 return LowerFP_TO_INT_SAT(Op, DAG);
1473 case ISD::CTTZ:
1475 case ISD::CTLZ:
1477 return LowerCTLZ_CTTZ(Op, DAG);
1478 case ISD::CTLS:
1479 return LowerCTLS(Op, DAG);
1481 }
1482 return Op;
1483}
1484
1487 SelectionDAG &DAG) const {
1488 switch (N->getOpcode()) {
1490 // Different parts of legalization seem to interpret which type of
1491 // sign_extend_inreg is the one to check for custom lowering. The extended
1492 // from type is what really matters, but some places check for custom
1493 // lowering of the result type. This results in trying to use
1494 // ReplaceNodeResults to sext_in_reg to an illegal type, so we'll just do
1495 // nothing here and let the illegal result integer be handled normally.
1496 return;
1497 case ISD::FLOG2:
1498 if (SDValue Lowered = LowerFLOG2(SDValue(N, 0), DAG))
1499 Results.push_back(Lowered);
1500 return;
1501 case ISD::FLOG:
1502 case ISD::FLOG10:
1503 if (SDValue Lowered = LowerFLOGCommon(SDValue(N, 0), DAG))
1504 Results.push_back(Lowered);
1505 return;
1506 case ISD::FEXP2:
1507 if (SDValue Lowered = lowerFEXP2(SDValue(N, 0), DAG))
1508 Results.push_back(Lowered);
1509 return;
1510 case ISD::FEXP:
1511 case ISD::FEXP10:
1512 if (SDValue Lowered = lowerFEXP(SDValue(N, 0), DAG))
1513 Results.push_back(Lowered);
1514 return;
1515 case ISD::CTLZ:
1517 if (auto Lowered = lowerCTLZResults(SDValue(N, 0u), DAG))
1518 Results.push_back(Lowered);
1519 return;
1520 default:
1521 return;
1522 }
1523}
1524
1526 SelectionDAG &DAG) const {
1528 SDLoc SL(Op);
1529 EVT VT = Op.getValueType();
1530 return DAG.getTargetBlockAddress(BA->getBlockAddress(), VT, BA->getOffset(),
1531 BA->getTargetFlags());
1532}
1533
1535 SDValue Op,
1536 SelectionDAG &DAG) const {
1537
1538 const DataLayout &DL = DAG.getDataLayout();
1540 const GlobalValue *GV = G->getGlobal();
1541
1542 if (!MFI->isModuleEntryFunction()) {
1543 bool IsNamedBarrier = AMDGPU::isNamedBarrier(*cast<GlobalVariable>(GV));
1544 std::optional<uint32_t> Address =
1546 if (!Address && IsNamedBarrier)
1547 llvm_unreachable("named barrier should have an assigned address");
1548 if (Address) {
1549 if (IsNamedBarrier) {
1550 unsigned BarCnt = cast<GlobalVariable>(GV)->getGlobalSize(DL) / 16;
1551 MFI->recordNumNamedBarriers(Address.value(), BarCnt);
1552 }
1553 // A constant byte offset (e.g. from a GEP into an array of named
1554 // barriers) folds directly into the fixed LDS address.
1555 return DAG.getConstant(*Address + G->getOffset(), SDLoc(Op),
1556 Op.getValueType());
1557 }
1558 }
1559
1560 if (G->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
1561 G->getAddressSpace() == AMDGPUAS::REGION_ADDRESS) {
1562 if (!MFI->isModuleEntryFunction() &&
1563 GV->getName() != "llvm.amdgcn.module.lds" &&
1565 SDLoc DL(Op);
1566 const Function &Fn = DAG.getMachineFunction().getFunction();
1568 Fn, "local memory global used by non-kernel function",
1569 DL.getDebugLoc(), DS_Warning));
1570
1571 // We currently don't have a way to correctly allocate LDS objects that
1572 // aren't directly associated with a kernel. We do force inlining of
1573 // functions that use local objects. However, if these dead functions are
1574 // not eliminated, we don't want a compile time error. Just emit a warning
1575 // and a trap, since there should be no callable path here.
1576 SDValue Trap = DAG.getNode(ISD::TRAP, DL, MVT::Other, DAG.getEntryNode());
1577 SDValue OutputChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
1578 Trap, DAG.getRoot());
1579 DAG.setRoot(OutputChain);
1580 return DAG.getPOISON(Op.getValueType());
1581 }
1582
1583 // TODO: We could emit code to handle the initialization somewhere.
1584 // We ignore the initializer for now and legalize it to allow selection.
1585 // The initializer will anyway get errored out during assembly emission.
1586 unsigned Offset = MFI->allocateLDSGlobal(DL, *cast<GlobalVariable>(GV));
1587 // A constant byte offset (e.g. from a GEP into an array of named barriers)
1588 // folds directly into the allocated LDS address.
1589 return DAG.getConstant(Offset + G->getOffset(), SDLoc(Op),
1590 Op.getValueType());
1591 }
1592 return SDValue();
1593}
1594
1596 SelectionDAG &DAG) const {
1598 SDLoc SL(Op);
1599
1600 EVT VT = Op.getValueType();
1601 if (VT.getVectorElementType().getSizeInBits() < 32) {
1602 unsigned OpBitSize = Op.getOperand(0).getValueType().getSizeInBits();
1603 if (OpBitSize >= 32 && OpBitSize % 32 == 0) {
1604 unsigned NewNumElt = OpBitSize / 32;
1605 EVT NewEltVT = (NewNumElt == 1) ? MVT::i32
1607 MVT::i32, NewNumElt);
1608 for (const SDUse &U : Op->ops()) {
1609 SDValue In = U.get();
1610 SDValue NewIn = DAG.getNode(ISD::BITCAST, SL, NewEltVT, In);
1611 if (NewNumElt > 1)
1612 DAG.ExtractVectorElements(NewIn, Args);
1613 else
1614 Args.push_back(NewIn);
1615 }
1616
1617 EVT NewVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32,
1618 NewNumElt * Op.getNumOperands());
1619 SDValue BV = DAG.getBuildVector(NewVT, SL, Args);
1620 return DAG.getNode(ISD::BITCAST, SL, VT, BV);
1621 }
1622 }
1623
1624 for (const SDUse &U : Op->ops())
1625 DAG.ExtractVectorElements(U.get(), Args);
1626
1627 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1628}
1629
1631 SelectionDAG &DAG) const {
1632 SDLoc SL(Op);
1634 unsigned Start = Op.getConstantOperandVal(1);
1635 EVT VT = Op.getValueType();
1636 EVT SrcVT = Op.getOperand(0).getValueType();
1637
1638 if (VT.getScalarSizeInBits() == 16 && Start % 2 == 0) {
1639 unsigned NumElt = VT.getVectorNumElements();
1640 unsigned NumSrcElt = SrcVT.getVectorNumElements();
1641 assert(NumElt % 2 == 0 && NumSrcElt % 2 == 0 && "expect legal types");
1642
1643 // Extract 32-bit registers at a time.
1644 EVT NewSrcVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumSrcElt / 2);
1645 EVT NewVT = NumElt == 2
1646 ? MVT::i32
1647 : EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElt / 2);
1648 SDValue Tmp = DAG.getNode(ISD::BITCAST, SL, NewSrcVT, Op.getOperand(0));
1649
1650 DAG.ExtractVectorElements(Tmp, Args, Start / 2, NumElt / 2);
1651 if (NumElt == 2)
1652 Tmp = Args[0];
1653 else
1654 Tmp = DAG.getBuildVector(NewVT, SL, Args);
1655
1656 return DAG.getNode(ISD::BITCAST, SL, VT, Tmp);
1657 }
1658
1659 DAG.ExtractVectorElements(Op.getOperand(0), Args, Start,
1661
1662 return DAG.getBuildVector(Op.getValueType(), SL, Args);
1663}
1664
1665// TODO: Handle fabs too
1667 if (Val.getOpcode() == ISD::FNEG)
1668 return Val.getOperand(0);
1669
1670 return Val;
1671}
1672
1674 if (Val.getOpcode() == ISD::FNEG)
1675 Val = Val.getOperand(0);
1676 if (Val.getOpcode() == ISD::FABS)
1677 Val = Val.getOperand(0);
1678 if (Val.getOpcode() == ISD::FCOPYSIGN)
1679 Val = Val.getOperand(0);
1680 return Val;
1681}
1682
1684 const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True,
1685 SDValue False, SDValue CC, DAGCombinerInfo &DCI) const {
1686 SelectionDAG &DAG = DCI.DAG;
1687 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(CC)->get();
1688 switch (CCOpcode) {
1689 case ISD::SETOEQ:
1690 case ISD::SETONE:
1691 case ISD::SETUNE:
1692 case ISD::SETNE:
1693 case ISD::SETUEQ:
1694 case ISD::SETEQ:
1695 case ISD::SETFALSE:
1696 case ISD::SETFALSE2:
1697 case ISD::SETTRUE:
1698 case ISD::SETTRUE2:
1699 case ISD::SETUO:
1700 case ISD::SETO:
1701 break;
1702 case ISD::SETULE:
1703 case ISD::SETULT: {
1704 if (LHS == True)
1705 return DAG.getNode(AMDGPUISD::FMIN_LEGACY, DL, VT, RHS, LHS);
1706 return DAG.getNode(AMDGPUISD::FMAX_LEGACY, DL, VT, LHS, RHS);
1707 }
1708 case ISD::SETOLE:
1709 case ISD::SETOLT:
1710 case ISD::SETLE:
1711 case ISD::SETLT: {
1712 // Ordered. Assume ordered for undefined.
1713
1714 // Only do this after legalization to avoid interfering with other combines
1715 // which might occur.
1717 !DCI.isCalledByLegalizer())
1718 return SDValue();
1719
1720 // We need to permute the operands to get the correct NaN behavior. The
1721 // selected operand is the second one based on the failing compare with NaN,
1722 // so permute it based on the compare type the hardware uses.
1723 if (LHS == True)
1724 return DAG.getNode(AMDGPUISD::FMIN_LEGACY, DL, VT, LHS, RHS);
1725 return DAG.getNode(AMDGPUISD::FMAX_LEGACY, DL, VT, RHS, LHS);
1726 }
1727 case ISD::SETUGE:
1728 case ISD::SETUGT: {
1729 if (LHS == True)
1730 return DAG.getNode(AMDGPUISD::FMAX_LEGACY, DL, VT, RHS, LHS);
1731 return DAG.getNode(AMDGPUISD::FMIN_LEGACY, DL, VT, LHS, RHS);
1732 }
1733 case ISD::SETGT:
1734 case ISD::SETGE:
1735 case ISD::SETOGE:
1736 case ISD::SETOGT: {
1738 !DCI.isCalledByLegalizer())
1739 return SDValue();
1740
1741 if (LHS == True)
1742 return DAG.getNode(AMDGPUISD::FMAX_LEGACY, DL, VT, LHS, RHS);
1743 return DAG.getNode(AMDGPUISD::FMIN_LEGACY, DL, VT, RHS, LHS);
1744 }
1745 case ISD::SETCC_INVALID:
1746 llvm_unreachable("Invalid setcc condcode!");
1747 }
1748 return SDValue();
1749}
1750
1751/// Generate Min/Max node
1753 SDValue LHS, SDValue RHS,
1754 SDValue True, SDValue False,
1755 SDValue CC,
1756 DAGCombinerInfo &DCI) const {
1757 if ((LHS == True && RHS == False) || (LHS == False && RHS == True))
1758 return combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, True, False, CC, DCI);
1759
1760 SelectionDAG &DAG = DCI.DAG;
1761
1762 // If we can't directly match this, try to see if we can fold an fneg to
1763 // match.
1764
1767 SDValue NegTrue = peekFNeg(True);
1768
1769 // Undo the combine foldFreeOpFromSelect does if it helps us match the
1770 // fmin/fmax.
1771 //
1772 // select (fcmp olt (lhs, K)), (fneg lhs), -K
1773 // -> fneg (fmin_legacy lhs, K)
1774 //
1775 // TODO: Use getNegatedExpression
1776 if (LHS == NegTrue && CFalse && CRHS) {
1777 APFloat NegRHS = neg(CRHS->getValueAPF());
1778 if (NegRHS == CFalse->getValueAPF()) {
1779 SDValue Combined =
1780 combineFMinMaxLegacyImpl(DL, VT, LHS, RHS, NegTrue, False, CC, DCI);
1781 if (Combined)
1782 return DAG.getNode(ISD::FNEG, DL, VT, Combined);
1783 return SDValue();
1784 }
1785 }
1786
1787 return SDValue();
1788}
1789
1790std::pair<SDValue, SDValue>
1792 SDLoc SL(Op);
1793
1794 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1795
1796 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1797 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1798
1799 SDValue Lo = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1800 SDValue Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1801
1802 return std::pair(Lo, Hi);
1803}
1804
1806 SDLoc SL(Op);
1807
1808 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1809 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
1810 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, Zero);
1811}
1812
1814 SDLoc SL(Op);
1815
1816 SDValue Vec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op);
1817 const SDValue One = DAG.getConstant(1, SL, MVT::i32);
1818 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Vec, One);
1819}
1820
1821// Split a vector type into two parts. The first part is a power of two vector.
1822// The second part is whatever is left over, and is a scalar if it would
1823// otherwise be a 1-vector.
1824std::pair<EVT, EVT>
1826 EVT LoVT, HiVT;
1827 EVT EltVT = VT.getVectorElementType();
1828 unsigned NumElts = VT.getVectorNumElements();
1829 unsigned LoNumElts = PowerOf2Ceil((NumElts + 1) / 2);
1830 LoVT = EVT::getVectorVT(*DAG.getContext(), EltVT, LoNumElts);
1831 HiVT = NumElts - LoNumElts == 1
1832 ? EltVT
1833 : EVT::getVectorVT(*DAG.getContext(), EltVT, NumElts - LoNumElts);
1834 return std::pair(LoVT, HiVT);
1835}
1836
1837// Split a vector value into two parts of types LoVT and HiVT. HiVT could be
1838// scalar.
1839std::pair<SDValue, SDValue>
1841 const EVT &LoVT, const EVT &HiVT,
1842 SelectionDAG &DAG) const {
1843 EVT VT = N.getValueType();
1845 (HiVT.isVector() ? HiVT.getVectorNumElements() : 1) <=
1846 VT.getVectorNumElements() &&
1847 "More vector elements requested than available!");
1849 DAG.getVectorIdxConstant(0, DL));
1850
1851 unsigned LoNumElts = LoVT.getVectorNumElements();
1852
1853 if (HiVT.isVector()) {
1854 unsigned HiNumElts = HiVT.getVectorNumElements();
1855 if ((VT.getVectorNumElements() % HiNumElts) == 0) {
1856 // Avoid creating an extract_subvector with an index that isn't a multiple
1857 // of the result type.
1859 DAG.getConstant(LoNumElts, DL, MVT::i32));
1860 return {Lo, Hi};
1861 }
1862
1864 DAG.ExtractVectorElements(N, Elts, /*Start=*/LoNumElts,
1865 /*Count=*/HiNumElts);
1866 SDValue Hi = DAG.getBuildVector(HiVT, DL, Elts);
1867 return {Lo, Hi};
1868 }
1869
1871 DAG.getVectorIdxConstant(LoNumElts, DL));
1872 return {Lo, Hi};
1873}
1874
1876 SelectionDAG &DAG) const {
1878 EVT VT = Op.getValueType();
1879 SDLoc SL(Op);
1880
1881
1882 // If this is a 2 element vector, we really want to scalarize and not create
1883 // weird 1 element vectors.
1884 if (VT.getVectorNumElements() == 2) {
1885 SDValue Ops[2];
1886 std::tie(Ops[0], Ops[1]) = scalarizeVectorLoad(Load, DAG);
1887 return DAG.getMergeValues(Ops, SL);
1888 }
1889
1890 SDValue BasePtr = Load->getBasePtr();
1891 EVT MemVT = Load->getMemoryVT();
1892
1893 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1894
1895 EVT LoVT, HiVT;
1896 EVT LoMemVT, HiMemVT;
1897 SDValue Lo, Hi;
1898
1899 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
1900 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
1901 std::tie(Lo, Hi) = splitVector(Op, SL, LoVT, HiVT, DAG);
1902
1903 unsigned Size = LoMemVT.getStoreSize();
1904 Align BaseAlign = Load->getAlign();
1905 Align HiAlign = commonAlignment(BaseAlign, Size);
1906
1907 SDValue LoLoad = DAG.getExtLoad(
1908 Load->getExtensionType(), SL, LoVT, Load->getChain(), BasePtr, SrcValue,
1909 LoMemVT, BaseAlign, Load->getMemOperand()->getFlags(), Load->getAAInfo());
1910 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, TypeSize::getFixed(Size));
1911 SDValue HiLoad = DAG.getExtLoad(
1912 Load->getExtensionType(), SL, HiVT, Load->getChain(), HiPtr,
1913 SrcValue.getWithOffset(LoMemVT.getStoreSize()), HiMemVT, HiAlign,
1914 Load->getMemOperand()->getFlags(), Load->getAAInfo());
1915
1916 SDValue Join;
1917 if (LoVT == HiVT) {
1918 // This is the case that the vector is power of two so was evenly split.
1919 Join = DAG.getNode(ISD::CONCAT_VECTORS, SL, VT, LoLoad, HiLoad);
1920 } else {
1921 Join = DAG.getNode(ISD::INSERT_SUBVECTOR, SL, VT, DAG.getPOISON(VT), LoLoad,
1922 DAG.getVectorIdxConstant(0, SL));
1923 Join = DAG.getNode(
1925 VT, Join, HiLoad,
1927 }
1928
1929 SDValue Ops[] = {Join, DAG.getNode(ISD::TokenFactor, SL, MVT::Other,
1930 LoLoad.getValue(1), HiLoad.getValue(1))};
1931
1932 return DAG.getMergeValues(Ops, SL);
1933}
1934
1936 SelectionDAG &DAG) const {
1938 EVT VT = Op.getValueType();
1939 SDValue BasePtr = Load->getBasePtr();
1940 EVT MemVT = Load->getMemoryVT();
1941 SDLoc SL(Op);
1942 const MachinePointerInfo &SrcValue = Load->getMemOperand()->getPointerInfo();
1943 Align BaseAlign = Load->getAlign();
1944 unsigned NumElements = MemVT.getVectorNumElements();
1945
1946 // Widen from vec3 to vec4 when the load is at least 8-byte aligned
1947 // or 16-byte fully dereferenceable. Otherwise, split the vector load.
1948 if (NumElements != 3 ||
1949 (BaseAlign < Align(8) &&
1950 !SrcValue.isDereferenceable(16, *DAG.getContext(), DAG.getDataLayout())))
1951 return SplitVectorLoad(Op, DAG);
1952
1953 assert(NumElements == 3);
1954
1955 EVT WideVT =
1957 EVT WideMemVT =
1959 SDValue WideLoad = DAG.getExtLoad(
1960 Load->getExtensionType(), SL, WideVT, Load->getChain(), BasePtr, SrcValue,
1961 WideMemVT, BaseAlign, Load->getMemOperand()->getFlags());
1962 return DAG.getMergeValues(
1963 {DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, VT, WideLoad,
1964 DAG.getVectorIdxConstant(0, SL)),
1965 WideLoad.getValue(1)},
1966 SL);
1967}
1968
1970 SelectionDAG &DAG) const {
1972 SDValue Val = Store->getValue();
1973 EVT VT = Val.getValueType();
1974
1975 // If this is a 2 element vector, we really want to scalarize and not create
1976 // weird 1 element vectors.
1977 if (VT.getVectorNumElements() == 2)
1978 return scalarizeVectorStore(Store, DAG);
1979
1980 EVT MemVT = Store->getMemoryVT();
1981 SDValue Chain = Store->getChain();
1982 SDValue BasePtr = Store->getBasePtr();
1983 SDLoc SL(Op);
1984
1985 EVT LoVT, HiVT;
1986 EVT LoMemVT, HiMemVT;
1987 SDValue Lo, Hi;
1988
1989 std::tie(LoVT, HiVT) = getSplitDestVTs(VT, DAG);
1990 std::tie(LoMemVT, HiMemVT) = getSplitDestVTs(MemVT, DAG);
1991 std::tie(Lo, Hi) = splitVector(Val, SL, LoVT, HiVT, DAG);
1992
1993 SDValue HiPtr = DAG.getObjectPtrOffset(SL, BasePtr, LoMemVT.getStoreSize());
1994
1995 const MachinePointerInfo &SrcValue = Store->getMemOperand()->getPointerInfo();
1996 Align BaseAlign = Store->getAlign();
1997 unsigned Size = LoMemVT.getStoreSize();
1998 Align HiAlign = commonAlignment(BaseAlign, Size);
1999
2000 SDValue LoStore =
2001 DAG.getTruncStore(Chain, SL, Lo, BasePtr, SrcValue, LoMemVT, BaseAlign,
2002 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2003 SDValue HiStore = DAG.getTruncStore(
2004 Chain, SL, Hi, HiPtr, SrcValue.getWithOffset(Size), HiMemVT, HiAlign,
2005 Store->getMemOperand()->getFlags(), Store->getAAInfo());
2006
2007 return DAG.getNode(ISD::TokenFactor, SL, MVT::Other, LoStore, HiStore);
2008}
2009
2010// This is a shortcut for integer division because we have fast i32<->f32
2011// conversions, and fast f32 reciprocal instructions.
2013 bool Sign) const {
2014 SDLoc DL(Op);
2015 EVT VT = Op.getValueType();
2016 assert(VT == MVT::i32 && "LowerDIVREMToFloat expects an i32");
2017
2018 SDValue LHS = Op.getOperand(0);
2019 SDValue RHS = Op.getOperand(1);
2020 MVT IntVT = MVT::i32;
2021 MVT FltVT = MVT::f32;
2022
2023 unsigned LHSSignBits;
2024 unsigned RHSSignBits;
2025 if (Sign) {
2026 LHSSignBits = DAG.ComputeNumSignBits(LHS);
2027 RHSSignBits = DAG.ComputeNumSignBits(RHS);
2028 if (LHSSignBits < 9 || RHSSignBits < 9)
2029 return SDValue();
2030 } else {
2031 KnownBits LHSKnown = DAG.computeKnownBits(LHS);
2032 KnownBits RHSKnown = DAG.computeKnownBits(RHS);
2033
2034 LHSSignBits = LHSKnown.countMinLeadingZeros();
2035 RHSSignBits = RHSKnown.countMinLeadingZeros();
2036 }
2037
2038 unsigned BitSize = VT.getSizeInBits();
2039 unsigned SignBits = std::min(LHSSignBits, RHSSignBits);
2040 unsigned DivBits = BitSize - SignBits;
2041 if (Sign)
2042 ++DivBits;
2043
2044 // In order to avoid problems due to 1 ulp accuracy issues with v_rcp_f32,
2045 // limit LowerDIVREMToFloat to:
2046 // [-0x400000,0x3FFFFF] for Sign
2047 // [ 0x000000,0x3FFFFF] for !Sign
2048 // This matches what is done in expandDivRemToFloatImpl.
2049 if (DivBits > (Sign ? 23 : 22))
2050 return SDValue();
2051
2054
2055 // int ia = (int)LHS;
2056 SDValue ia = LHS;
2057
2058 // int ib, (int)RHS;
2059 SDValue ib = RHS;
2060
2061 // The calculation:
2062 // fq = fa*recip(fb)
2063 // may be too small due to the 1ulp accuracy in the recip
2064 // operation and rounding issues. Since fq is truncated to produce
2065 // an integer value it may be too small by one. This is
2066 // dealt with by incrementing fa by 1ulp:
2067 // fq = (fa+1ulp)*recip(fb)
2068 // This will increase fa's magnitude by at most 0.5
2069 // (i.e. when fabs(fa)==0x400000 the LSB of the mantissa represents 0.5).
2070 // Thus, this method is safe since fa must be incremented by at least 1.0
2071 // for the quotient to increase by one.
2072 SDValue fa = DAG.getNode(ToFp, DL, FltVT, ia);
2073 SDValue faAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, fa);
2074 SDValue faIncremented = DAG.getNode(ISD::ADD, DL, MVT::i32, faAsInt,
2075 DAG.getConstant(1, DL, MVT::i32));
2076 fa = DAG.getNode(ISD::BITCAST, DL, FltVT, faIncremented);
2077
2078 // float fb = (float)ib;
2079 SDValue fb = DAG.getNode(ToFp, DL, FltVT, ib);
2080
2081 SDValue fq = DAG.getNode(ISD::FMUL, DL, FltVT,
2082 fa, DAG.getNode(AMDGPUISD::RCP, DL, FltVT, fb));
2083
2084 // fq = trunc(fq);
2085 fq = DAG.getNode(ISD::FTRUNC, DL, FltVT, fq);
2086
2087 // int iq = (int)fq;
2088 SDValue Div = DAG.getNode(ToInt, DL, IntVT, fq);
2089
2090 // Rem needs compensation, it's easier to recompute it
2091 SDValue Rem = DAG.getNode(ISD::MUL, DL, VT, Div, RHS);
2092 Rem = DAG.getNode(ISD::SUB, DL, VT, LHS, Rem);
2093
2094 return DAG.getMergeValues({ Div, Rem }, DL);
2095}
2096
2098 SelectionDAG &DAG,
2100 SDLoc DL(Op);
2101 EVT VT = Op.getValueType();
2102
2103 assert(VT == MVT::i64 && "LowerUDIVREM64 expects an i64");
2104
2105 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2106
2107 SDValue One = DAG.getConstant(1, DL, HalfVT);
2108 SDValue Zero = DAG.getConstant(0, DL, HalfVT);
2109
2110 //HiLo split
2111 SDValue LHS_Lo, LHS_Hi;
2112 SDValue LHS = Op.getOperand(0);
2113 std::tie(LHS_Lo, LHS_Hi) = DAG.SplitScalar(LHS, DL, HalfVT, HalfVT);
2114
2115 SDValue RHS_Lo, RHS_Hi;
2116 SDValue RHS = Op.getOperand(1);
2117 std::tie(RHS_Lo, RHS_Hi) = DAG.SplitScalar(RHS, DL, HalfVT, HalfVT);
2118
2119 if (DAG.MaskedValueIsZero(RHS, APInt::getHighBitsSet(64, 32)) &&
2120 DAG.MaskedValueIsZero(LHS, APInt::getHighBitsSet(64, 32))) {
2121
2122 SDValue Res = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2123 LHS_Lo, RHS_Lo);
2124
2125 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(0), Zero});
2126 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {Res.getValue(1), Zero});
2127
2128 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV));
2129 Results.push_back(DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM));
2130 return;
2131 }
2132
2133 if (isTypeLegal(MVT::i64)) {
2134 // The algorithm here is based on ideas from "Software Integer Division",
2135 // Tom Rodeheffer, August 2008.
2136
2139
2140 // Compute denominator reciprocal.
2141 unsigned FMAD =
2142 !Subtarget->hasMadMacF32Insts() ? (unsigned)ISD::FMA
2145 : (unsigned)AMDGPUISD::FMAD_FTZ;
2146
2147 SDValue Cvt_Lo = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Lo);
2148 SDValue Cvt_Hi = DAG.getNode(ISD::UINT_TO_FP, DL, MVT::f32, RHS_Hi);
2149 SDValue Mad1 = DAG.getNode(FMAD, DL, MVT::f32, Cvt_Hi,
2150 DAG.getConstantFP(APInt(32, 0x4f800000).bitsToFloat(), DL, MVT::f32),
2151 Cvt_Lo);
2152 SDValue Rcp = DAG.getNode(AMDGPUISD::RCP, DL, MVT::f32, Mad1);
2153 SDValue Mul1 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Rcp,
2154 DAG.getConstantFP(APInt(32, 0x5f7ffffc).bitsToFloat(), DL, MVT::f32));
2155 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f32, Mul1,
2156 DAG.getConstantFP(APInt(32, 0x2f800000).bitsToFloat(), DL, MVT::f32));
2157 SDValue Trunc = DAG.getNode(ISD::FTRUNC, DL, MVT::f32, Mul2);
2158 SDValue Mad2 = DAG.getNode(FMAD, DL, MVT::f32, Trunc,
2159 DAG.getConstantFP(APInt(32, 0xcf800000).bitsToFloat(), DL, MVT::f32),
2160 Mul1);
2161 SDValue Rcp_Lo = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Mad2);
2162 SDValue Rcp_Hi = DAG.getNode(ISD::FP_TO_UINT, DL, HalfVT, Trunc);
2163 SDValue Rcp64 = DAG.getBitcast(VT,
2164 DAG.getBuildVector(MVT::v2i32, DL, {Rcp_Lo, Rcp_Hi}));
2165
2166 SDValue Zero64 = DAG.getConstant(0, DL, VT);
2167 SDValue One64 = DAG.getConstant(1, DL, VT);
2168 SDValue Zero1 = DAG.getConstant(0, DL, MVT::i1);
2169 SDVTList HalfCarryVT = DAG.getVTList(HalfVT, MVT::i1);
2170
2171 // First round of UNR (Unsigned integer Newton-Raphson).
2172 SDValue Neg_RHS = DAG.getNode(ISD::SUB, DL, VT, Zero64, RHS);
2173 SDValue Mullo1 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Rcp64);
2174 SDValue Mulhi1 = DAG.getNode(ISD::MULHU, DL, VT, Rcp64, Mullo1);
2175 SDValue Mulhi1_Lo, Mulhi1_Hi;
2176 std::tie(Mulhi1_Lo, Mulhi1_Hi) =
2177 DAG.SplitScalar(Mulhi1, DL, HalfVT, HalfVT);
2178 SDValue Add1_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Lo,
2179 Mulhi1_Lo, Zero1);
2180 SDValue Add1_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Rcp_Hi,
2181 Mulhi1_Hi, Add1_Lo.getValue(1));
2182 SDValue Add1 = DAG.getBitcast(VT,
2183 DAG.getBuildVector(MVT::v2i32, DL, {Add1_Lo, Add1_Hi}));
2184
2185 // Second round of UNR.
2186 SDValue Mullo2 = DAG.getNode(ISD::MUL, DL, VT, Neg_RHS, Add1);
2187 SDValue Mulhi2 = DAG.getNode(ISD::MULHU, DL, VT, Add1, Mullo2);
2188 SDValue Mulhi2_Lo, Mulhi2_Hi;
2189 std::tie(Mulhi2_Lo, Mulhi2_Hi) =
2190 DAG.SplitScalar(Mulhi2, DL, HalfVT, HalfVT);
2191 SDValue Add2_Lo = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Lo,
2192 Mulhi2_Lo, Zero1);
2193 SDValue Add2_Hi = DAG.getNode(ISD::UADDO_CARRY, DL, HalfCarryVT, Add1_Hi,
2194 Mulhi2_Hi, Add2_Lo.getValue(1));
2195 SDValue Add2 = DAG.getBitcast(VT,
2196 DAG.getBuildVector(MVT::v2i32, DL, {Add2_Lo, Add2_Hi}));
2197
2198 SDValue Mulhi3 = DAG.getNode(ISD::MULHU, DL, VT, LHS, Add2);
2199
2200 SDValue Mul3 = DAG.getNode(ISD::MUL, DL, VT, RHS, Mulhi3);
2201
2202 SDValue Mul3_Lo, Mul3_Hi;
2203 std::tie(Mul3_Lo, Mul3_Hi) = DAG.SplitScalar(Mul3, DL, HalfVT, HalfVT);
2204 SDValue Sub1_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Lo,
2205 Mul3_Lo, Zero1);
2206 SDValue Sub1_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, LHS_Hi,
2207 Mul3_Hi, Sub1_Lo.getValue(1));
2208 SDValue Sub1_Mi = DAG.getNode(ISD::SUB, DL, HalfVT, LHS_Hi, Mul3_Hi);
2209 SDValue Sub1 = DAG.getBitcast(VT,
2210 DAG.getBuildVector(MVT::v2i32, DL, {Sub1_Lo, Sub1_Hi}));
2211
2212 SDValue MinusOne = DAG.getConstant(0xffffffffu, DL, HalfVT);
2213 SDValue C1 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, MinusOne, Zero,
2214 ISD::SETUGE);
2215 SDValue C2 = DAG.getSelectCC(DL, Sub1_Lo, RHS_Lo, MinusOne, Zero,
2216 ISD::SETUGE);
2217 SDValue C3 = DAG.getSelectCC(DL, Sub1_Hi, RHS_Hi, C2, C1, ISD::SETEQ);
2218
2219 // TODO: Here and below portions of the code can be enclosed into if/endif.
2220 // Currently control flow is unconditional and we have 4 selects after
2221 // potential endif to substitute PHIs.
2222
2223 // if C3 != 0 ...
2224 SDValue Sub2_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Lo,
2225 RHS_Lo, Zero1);
2226 SDValue Sub2_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub1_Mi,
2227 RHS_Hi, Sub1_Lo.getValue(1));
2228 SDValue Sub2_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2229 Zero, Sub2_Lo.getValue(1));
2230 SDValue Sub2 = DAG.getBitcast(VT,
2231 DAG.getBuildVector(MVT::v2i32, DL, {Sub2_Lo, Sub2_Hi}));
2232
2233 SDValue Add3 = DAG.getNode(ISD::ADD, DL, VT, Mulhi3, One64);
2234
2235 SDValue C4 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, MinusOne, Zero,
2236 ISD::SETUGE);
2237 SDValue C5 = DAG.getSelectCC(DL, Sub2_Lo, RHS_Lo, MinusOne, Zero,
2238 ISD::SETUGE);
2239 SDValue C6 = DAG.getSelectCC(DL, Sub2_Hi, RHS_Hi, C5, C4, ISD::SETEQ);
2240
2241 // if (C6 != 0)
2242 SDValue Add4 = DAG.getNode(ISD::ADD, DL, VT, Add3, One64);
2243
2244 SDValue Sub3_Lo = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Lo,
2245 RHS_Lo, Zero1);
2246 SDValue Sub3_Mi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub2_Mi,
2247 RHS_Hi, Sub2_Lo.getValue(1));
2248 SDValue Sub3_Hi = DAG.getNode(ISD::USUBO_CARRY, DL, HalfCarryVT, Sub3_Mi,
2249 Zero, Sub3_Lo.getValue(1));
2250 SDValue Sub3 = DAG.getBitcast(VT,
2251 DAG.getBuildVector(MVT::v2i32, DL, {Sub3_Lo, Sub3_Hi}));
2252
2253 // endif C6
2254 // endif C3
2255
2256 SDValue Sel1 = DAG.getSelectCC(DL, C6, Zero, Add4, Add3, ISD::SETNE);
2257 SDValue Div = DAG.getSelectCC(DL, C3, Zero, Sel1, Mulhi3, ISD::SETNE);
2258
2259 SDValue Sel2 = DAG.getSelectCC(DL, C6, Zero, Sub3, Sub2, ISD::SETNE);
2260 SDValue Rem = DAG.getSelectCC(DL, C3, Zero, Sel2, Sub1, ISD::SETNE);
2261
2262 Results.push_back(Div);
2263 Results.push_back(Rem);
2264
2265 return;
2266 }
2267
2268 // r600 expandion.
2269 // Get Speculative values
2270 SDValue DIV_Part = DAG.getNode(ISD::UDIV, DL, HalfVT, LHS_Hi, RHS_Lo);
2271 SDValue REM_Part = DAG.getNode(ISD::UREM, DL, HalfVT, LHS_Hi, RHS_Lo);
2272
2273 SDValue REM_Lo = DAG.getSelectCC(DL, RHS_Hi, Zero, REM_Part, LHS_Hi, ISD::SETEQ);
2274 SDValue REM = DAG.getBuildVector(MVT::v2i32, DL, {REM_Lo, Zero});
2275 REM = DAG.getNode(ISD::BITCAST, DL, MVT::i64, REM);
2276
2277 SDValue DIV_Hi = DAG.getSelectCC(DL, RHS_Hi, Zero, DIV_Part, Zero, ISD::SETEQ);
2278 SDValue DIV_Lo = Zero;
2279
2280 const unsigned halfBitWidth = HalfVT.getSizeInBits();
2281
2282 for (unsigned i = 0; i < halfBitWidth; ++i) {
2283 const unsigned bitPos = halfBitWidth - i - 1;
2284 SDValue POS = DAG.getConstant(bitPos, DL, HalfVT);
2285 // Get value of high bit
2286 SDValue HBit = DAG.getNode(ISD::SRL, DL, HalfVT, LHS_Lo, POS);
2287 HBit = DAG.getNode(ISD::AND, DL, HalfVT, HBit, One);
2288 HBit = DAG.getNode(ISD::ZERO_EXTEND, DL, VT, HBit);
2289
2290 // Shift
2291 REM = DAG.getNode(ISD::SHL, DL, VT, REM, DAG.getConstant(1, DL, VT));
2292 // Add LHS high bit
2293 REM = DAG.getNode(ISD::OR, DL, VT, REM, HBit);
2294
2295 SDValue BIT = DAG.getConstant(1ULL << bitPos, DL, HalfVT);
2296 SDValue realBIT = DAG.getSelectCC(DL, REM, RHS, BIT, Zero, ISD::SETUGE);
2297
2298 DIV_Lo = DAG.getNode(ISD::OR, DL, HalfVT, DIV_Lo, realBIT);
2299
2300 // Update REM
2301 SDValue REM_sub = DAG.getNode(ISD::SUB, DL, VT, REM, RHS);
2302 REM = DAG.getSelectCC(DL, REM, RHS, REM_sub, REM, ISD::SETUGE);
2303 }
2304
2305 SDValue DIV = DAG.getBuildVector(MVT::v2i32, DL, {DIV_Lo, DIV_Hi});
2306 DIV = DAG.getNode(ISD::BITCAST, DL, MVT::i64, DIV);
2307 Results.push_back(DIV);
2308 Results.push_back(REM);
2309}
2310
2312 SelectionDAG &DAG) const {
2313 SDLoc DL(Op);
2314 EVT VT = Op.getValueType();
2315
2316 if (VT == MVT::i64) {
2318 LowerUDIVREM64(Op, DAG, Results);
2319 return DAG.getMergeValues(Results, DL);
2320 }
2321
2322 if (VT == MVT::i32) {
2323 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, false))
2324 return Res;
2325 }
2326
2327 SDValue X = Op.getOperand(0);
2328 SDValue Y = Op.getOperand(1);
2329
2330 // See AMDGPUCodeGenPrepare::expandDivRem32 for a description of the
2331 // algorithm used here.
2332
2333 // Initial estimate of inv(y).
2334 SDValue Z = DAG.getNode(AMDGPUISD::URECIP, DL, VT, Y);
2335
2336 // One round of UNR.
2337 SDValue NegY = DAG.getNode(ISD::SUB, DL, VT, DAG.getConstant(0, DL, VT), Y);
2338 SDValue NegYZ = DAG.getNode(ISD::MUL, DL, VT, NegY, Z);
2339 Z = DAG.getNode(ISD::ADD, DL, VT, Z,
2340 DAG.getNode(ISD::MULHU, DL, VT, Z, NegYZ));
2341
2342 // Quotient/remainder estimate.
2343 SDValue Q = DAG.getNode(ISD::MULHU, DL, VT, X, Z);
2344 SDValue R =
2345 DAG.getNode(ISD::SUB, DL, VT, X, DAG.getNode(ISD::MUL, DL, VT, Q, Y));
2346
2347 // First quotient/remainder refinement.
2348 EVT CCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2349 SDValue One = DAG.getConstant(1, DL, VT);
2350 SDValue Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2351 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2352 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2353 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2354 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2355
2356 // Second quotient/remainder refinement.
2357 Cond = DAG.getSetCC(DL, CCVT, R, Y, ISD::SETUGE);
2358 Q = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2359 DAG.getNode(ISD::ADD, DL, VT, Q, One), Q);
2360 R = DAG.getNode(ISD::SELECT, DL, VT, Cond,
2361 DAG.getNode(ISD::SUB, DL, VT, R, Y), R);
2362
2363 return DAG.getMergeValues({Q, R}, DL);
2364}
2365
2367 SelectionDAG &DAG) const {
2368 SDLoc DL(Op);
2369 EVT VT = Op.getValueType();
2370
2371 SDValue LHS = Op.getOperand(0);
2372 SDValue RHS = Op.getOperand(1);
2373
2374 SDValue Zero = DAG.getConstant(0, DL, VT);
2375 SDValue NegOne = DAG.getAllOnesConstant(DL, VT);
2376
2377 if (VT == MVT::i32) {
2378 if (SDValue Res = LowerDIVREMToFloat(Op, DAG, true))
2379 return Res;
2380 }
2381
2382 // LHS must have > 33 sign-bits to ensure that LHS != -2147483648
2383 // Otherwise 32-bit division cannot be used safely.
2384 // -2147483648/1 and -2147483648/-1 are not equal,
2385 // but they produce the same lower 32-bit result.
2386 if (VT == MVT::i64 && DAG.ComputeNumSignBits(LHS) > 33 &&
2387 DAG.ComputeNumSignBits(RHS) > 32) {
2388 EVT HalfVT = VT.getHalfSizedIntegerVT(*DAG.getContext());
2389
2390 //HiLo split
2391 SDValue LHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, LHS, Zero);
2392 SDValue RHS_Lo = DAG.getNode(ISD::EXTRACT_ELEMENT, DL, HalfVT, RHS, Zero);
2393 SDValue DIVREM = DAG.getNode(ISD::SDIVREM, DL, DAG.getVTList(HalfVT, HalfVT),
2394 LHS_Lo, RHS_Lo);
2395 SDValue Res[2] = {
2396 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(0)),
2397 DAG.getNode(ISD::SIGN_EXTEND, DL, VT, DIVREM.getValue(1))
2398 };
2399 return DAG.getMergeValues(Res, DL);
2400 }
2401
2402 SDValue LHSign = DAG.getSelectCC(DL, LHS, Zero, NegOne, Zero, ISD::SETLT);
2403 SDValue RHSign = DAG.getSelectCC(DL, RHS, Zero, NegOne, Zero, ISD::SETLT);
2404 SDValue DSign = DAG.getNode(ISD::XOR, DL, VT, LHSign, RHSign);
2405 SDValue RSign = LHSign; // Remainder sign is the same as LHS
2406
2407 LHS = DAG.getNode(ISD::ADD, DL, VT, LHS, LHSign);
2408 RHS = DAG.getNode(ISD::ADD, DL, VT, RHS, RHSign);
2409
2410 LHS = DAG.getNode(ISD::XOR, DL, VT, LHS, LHSign);
2411 RHS = DAG.getNode(ISD::XOR, DL, VT, RHS, RHSign);
2412
2413 SDValue Div = DAG.getNode(ISD::UDIVREM, DL, DAG.getVTList(VT, VT), LHS, RHS);
2414 SDValue Rem = Div.getValue(1);
2415
2416 Div = DAG.getNode(ISD::XOR, DL, VT, Div, DSign);
2417 Rem = DAG.getNode(ISD::XOR, DL, VT, Rem, RSign);
2418
2419 Div = DAG.getNode(ISD::SUB, DL, VT, Div, DSign);
2420 Rem = DAG.getNode(ISD::SUB, DL, VT, Rem, RSign);
2421
2422 SDValue Res[2] = {
2423 Div,
2424 Rem
2425 };
2426 return DAG.getMergeValues(Res, DL);
2427}
2428
2430 SDLoc SL(Op);
2431 SDValue Src = Op.getOperand(0);
2432
2433 // result = trunc(src)
2434 // if (src > 0.0 && src != result)
2435 // result += 1.0
2436
2437 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2438
2439 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2440 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f64);
2441
2442 EVT SetCCVT =
2443 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2444
2445 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOGT);
2446 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2447 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2448
2449 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, One, Zero);
2450 // TODO: Should this propagate fast-math-flags?
2451 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2452}
2453
2455 SelectionDAG &DAG) {
2456 const unsigned FractBits = 52;
2457 const unsigned ExpBits = 11;
2458
2459 SDValue ExpPart = DAG.getNode(AMDGPUISD::BFE_U32, SL, MVT::i32,
2460 Hi,
2461 DAG.getConstant(FractBits - 32, SL, MVT::i32),
2462 DAG.getConstant(ExpBits, SL, MVT::i32));
2463 SDValue Exp = DAG.getNode(ISD::SUB, SL, MVT::i32, ExpPart,
2464 DAG.getConstant(1023, SL, MVT::i32));
2465
2466 return Exp;
2467}
2468
2470 SDLoc SL(Op);
2471 SDValue Src = Op.getOperand(0);
2472
2473 assert(Op.getValueType() == MVT::f64);
2474
2475 const SDValue Zero = DAG.getConstant(0, SL, MVT::i32);
2476
2477 // Extract the upper half, since this is where we will find the sign and
2478 // exponent.
2479 SDValue Hi = getHiHalf64(Src, DAG);
2480
2481 SDValue Exp = extractF64Exponent(Hi, SL, DAG);
2482
2483 const unsigned FractBits = 52;
2484
2485 // Extract the sign bit.
2486 const SDValue SignBitMask = DAG.getConstant(UINT32_C(1) << 31, SL, MVT::i32);
2487 SDValue SignBit = DAG.getNode(ISD::AND, SL, MVT::i32, Hi, SignBitMask);
2488
2489 // Extend back to 64-bits.
2490 SDValue SignBit64 = DAG.getBuildVector(MVT::v2i32, SL, {Zero, SignBit});
2491 SignBit64 = DAG.getNode(ISD::BITCAST, SL, MVT::i64, SignBit64);
2492
2493 SDValue BcInt = DAG.getNode(ISD::BITCAST, SL, MVT::i64, Src);
2494 const SDValue FractMask
2495 = DAG.getConstant((UINT64_C(1) << FractBits) - 1, SL, MVT::i64);
2496
2497 SDValue Shr = DAG.getNode(ISD::SRA, SL, MVT::i64, FractMask, Exp);
2498 SDValue Not = DAG.getNOT(SL, Shr, MVT::i64);
2499 SDValue Tmp0 = DAG.getNode(ISD::AND, SL, MVT::i64, BcInt, Not);
2500
2501 EVT SetCCVT =
2502 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::i32);
2503
2504 const SDValue FiftyOne = DAG.getConstant(FractBits - 1, SL, MVT::i32);
2505
2506 SDValue ExpLt0 = DAG.getSetCC(SL, SetCCVT, Exp, Zero, ISD::SETLT);
2507 SDValue ExpGt51 = DAG.getSetCC(SL, SetCCVT, Exp, FiftyOne, ISD::SETGT);
2508
2509 SDValue Tmp1 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpLt0, SignBit64, Tmp0);
2510 SDValue Tmp2 = DAG.getNode(ISD::SELECT, SL, MVT::i64, ExpGt51, BcInt, Tmp1);
2511
2512 return DAG.getNode(ISD::BITCAST, SL, MVT::f64, Tmp2);
2513}
2514
2516 SelectionDAG &DAG) const {
2517 SDLoc SL(Op);
2518 SDValue Src = Op.getOperand(0);
2519
2520 assert(Op.getValueType() == MVT::f64);
2521
2522 APFloat C1Val(APFloat::IEEEdouble(), "0x1.0p+52");
2523 SDValue C1 = DAG.getConstantFP(C1Val, SL, MVT::f64);
2524 SDValue CopySign = DAG.getNode(ISD::FCOPYSIGN, SL, MVT::f64, C1, Src);
2525
2526 // TODO: Should this propagate fast-math-flags?
2527
2528 SDValue Tmp1 = DAG.getNode(ISD::FADD, SL, MVT::f64, Src, CopySign);
2529 SDValue Tmp2 = DAG.getNode(ISD::FSUB, SL, MVT::f64, Tmp1, CopySign);
2530
2531 SDValue Fabs = DAG.getNode(ISD::FABS, SL, MVT::f64, Src);
2532
2533 APFloat C2Val(APFloat::IEEEdouble(), "0x1.fffffffffffffp+51");
2534 SDValue C2 = DAG.getConstantFP(C2Val, SL, MVT::f64);
2535
2536 EVT SetCCVT =
2537 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2538 SDValue Cond = DAG.getSetCC(SL, SetCCVT, Fabs, C2, ISD::SETOGT);
2539
2540 return DAG.getSelect(SL, MVT::f64, Cond, Src, Tmp2);
2541}
2542
2544 SelectionDAG &DAG) const {
2545 // FNEARBYINT and FRINT are the same, except in their handling of FP
2546 // exceptions. Those aren't really meaningful for us, and OpenCL only has
2547 // rint, so just treat them as equivalent.
2548 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), Op.getValueType(),
2549 Op.getOperand(0));
2550}
2551
2553 auto VT = Op.getValueType();
2554 auto Arg = Op.getOperand(0u);
2555 return DAG.getNode(ISD::FROUNDEVEN, SDLoc(Op), VT, Arg);
2556}
2557
2558// XXX - May require not supporting f32 denormals?
2559
2560// Don't handle v2f16. The extra instructions to scalarize and repack around the
2561// compare and vselect end up producing worse code than scalarizing the whole
2562// operation.
2564 SDLoc SL(Op);
2565 SDValue X = Op.getOperand(0);
2566 EVT VT = Op.getValueType();
2567
2568 SDValue T = DAG.getNode(ISD::FTRUNC, SL, VT, X);
2569
2570 // TODO: Should this propagate fast-math-flags?
2571
2572 SDValue Diff = DAG.getNode(ISD::FSUB, SL, VT, X, T);
2573
2574 SDValue AbsDiff = DAG.getNode(ISD::FABS, SL, VT, Diff);
2575
2576 const SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2577 const SDValue One = DAG.getConstantFP(1.0, SL, VT);
2578
2579 EVT SetCCVT =
2580 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
2581
2582 const SDValue Half = DAG.getConstantFP(0.5, SL, VT);
2583 SDValue Cmp = DAG.getSetCC(SL, SetCCVT, AbsDiff, Half, ISD::SETOGE);
2584 SDValue OneOrZeroFP = DAG.getNode(ISD::SELECT, SL, VT, Cmp, One, Zero);
2585
2586 SDValue SignedOffset = DAG.getNode(ISD::FCOPYSIGN, SL, VT, OneOrZeroFP, X);
2587 return DAG.getNode(ISD::FADD, SL, VT, T, SignedOffset);
2588}
2589
2591 SDLoc SL(Op);
2592 SDValue Src = Op.getOperand(0);
2593
2594 // result = trunc(src);
2595 // if (src < 0.0 && src != result)
2596 // result += -1.0.
2597
2598 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, MVT::f64, Src);
2599
2600 const SDValue Zero = DAG.getConstantFP(0.0, SL, MVT::f64);
2601 const SDValue NegOne = DAG.getConstantFP(-1.0, SL, MVT::f64);
2602
2603 EVT SetCCVT =
2604 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f64);
2605
2606 SDValue Lt0 = DAG.getSetCC(SL, SetCCVT, Src, Zero, ISD::SETOLT);
2607 SDValue NeTrunc = DAG.getSetCC(SL, SetCCVT, Src, Trunc, ISD::SETONE);
2608 SDValue And = DAG.getNode(ISD::AND, SL, SetCCVT, Lt0, NeTrunc);
2609
2610 SDValue Add = DAG.getNode(ISD::SELECT, SL, MVT::f64, And, NegOne, Zero);
2611 // TODO: Should this propagate fast-math-flags?
2612 return DAG.getNode(ISD::FADD, SL, MVT::f64, Trunc, Add);
2613}
2614
2615/// Return true if it's known that \p Src can never be an f32 denormal value.
2617 switch (Src.getOpcode()) {
2618 case ISD::FP_EXTEND:
2619 return Src.getOperand(0).getValueType() == MVT::f16;
2620 case ISD::FP16_TO_FP:
2621 case ISD::FFREXP:
2622 case ISD::FSQRT:
2623 case AMDGPUISD::LOG:
2624 case AMDGPUISD::EXP:
2625 return true;
2627 unsigned IntrinsicID = Src.getConstantOperandVal(0);
2628 switch (IntrinsicID) {
2629 case Intrinsic::amdgcn_frexp_mant:
2630 case Intrinsic::amdgcn_log:
2631 case Intrinsic::amdgcn_log_clamp:
2632 case Intrinsic::amdgcn_exp2:
2633 case Intrinsic::amdgcn_sqrt:
2634 return true;
2635 default:
2636 return false;
2637 }
2638 }
2639 default:
2640 return false;
2641 }
2642
2643 llvm_unreachable("covered opcode switch");
2644}
2645
2647 SDNodeFlags Flags) {
2648 return Flags.hasApproximateFuncs();
2649}
2650
2659
2661 SDValue Src,
2662 SDNodeFlags Flags) const {
2663 SDLoc SL(Src);
2664 EVT VT = Src.getValueType();
2665 const fltSemantics &Semantics = VT.getFltSemantics();
2666 SDValue SmallestNormal =
2667 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2668
2669 // Want to scale denormals up, but negatives and 0 work just as well on the
2670 // scaled path.
2671 SDValue IsLtSmallestNormal = DAG.getSetCC(
2672 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2673 SmallestNormal, ISD::SETOLT);
2674
2675 return IsLtSmallestNormal;
2676}
2677
2679 SDNodeFlags Flags) const {
2680 SDLoc SL(Src);
2681 EVT VT = Src.getValueType();
2682 const fltSemantics &Semantics = VT.getFltSemantics();
2683 SDValue Inf = DAG.getConstantFP(APFloat::getInf(Semantics), SL, VT);
2684
2685 SDValue Fabs = DAG.getNode(ISD::FABS, SL, VT, Src, Flags);
2686 SDValue IsFinite = DAG.getSetCC(
2687 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Fabs,
2688 Inf, ISD::SETOLT);
2689 return IsFinite;
2690}
2691
2692/// If denormal handling is required return the scaled input to FLOG2, and the
2693/// check for denormal range. Otherwise, return null values.
2694std::pair<SDValue, SDValue>
2696 SDValue Src, SDNodeFlags Flags) const {
2697 if (!needsDenormHandlingF32(DAG, Src, Flags))
2698 return {};
2699
2700 MVT VT = MVT::f32;
2701 const fltSemantics &Semantics = APFloat::IEEEsingle();
2702 SDValue SmallestNormal =
2703 DAG.getConstantFP(APFloat::getSmallestNormalized(Semantics), SL, VT);
2704
2705 SDValue IsLtSmallestNormal = DAG.getSetCC(
2706 SL, getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT), Src,
2707 SmallestNormal, ISD::SETOLT);
2708
2709 SDValue Scale32 = DAG.getConstantFP(0x1.0p+32, SL, VT);
2710 SDValue One = DAG.getConstantFP(1.0, SL, VT);
2711 SDValue ScaleFactor =
2712 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, Scale32, One, Flags);
2713
2714 SDValue ScaledInput = DAG.getNode(ISD::FMUL, SL, VT, Src, ScaleFactor, Flags);
2715 return {ScaledInput, IsLtSmallestNormal};
2716}
2717
2719 // v_log_f32 is good enough for OpenCL, except it doesn't handle denormals.
2720 // If we have to handle denormals, scale up the input and adjust the result.
2721
2722 // scaled = x * (is_denormal ? 0x1.0p+32 : 1.0)
2723 // log2 = amdgpu_log2 - (is_denormal ? 32.0 : 0.0)
2724
2725 SDLoc SL(Op);
2726 EVT VT = Op.getValueType();
2727 SDValue Src = Op.getOperand(0);
2728 SDNodeFlags Flags = Op->getFlags();
2729
2730 if (VT == MVT::f16) {
2731 // Nothing in half is a denormal when promoted to f32.
2732 assert(!isTypeLegal(VT));
2733 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
2734 SDValue Log = DAG.getNode(AMDGPUISD::LOG, SL, MVT::f32, Ext, Flags);
2735 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
2736 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
2737 }
2738
2739 auto [ScaledInput, IsLtSmallestNormal] =
2740 getScaledLogInput(DAG, SL, Src, Flags);
2741 if (!ScaledInput)
2742 return DAG.getNode(AMDGPUISD::LOG, SL, VT, Src, Flags);
2743
2744 SDValue Log2 = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2745
2746 SDValue ThirtyTwo = DAG.getConstantFP(32.0, SL, VT);
2747 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
2748 SDValue ResultOffset =
2749 DAG.getNode(ISD::SELECT, SL, VT, IsLtSmallestNormal, ThirtyTwo, Zero);
2750 return DAG.getNode(ISD::FSUB, SL, VT, Log2, ResultOffset, Flags);
2751}
2752
2753static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X,
2754 SDValue Y, SDValue C, SDNodeFlags Flags = SDNodeFlags()) {
2755 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Y, Flags);
2756 return DAG.getNode(ISD::FADD, SL, VT, Mul, C, Flags);
2757}
2758
2760 SelectionDAG &DAG) const {
2761 SDValue X = Op.getOperand(0);
2762 EVT VT = Op.getValueType();
2763 SDNodeFlags Flags = Op->getFlags();
2764 SDLoc DL(Op);
2765 const bool IsLog10 = Op.getOpcode() == ISD::FLOG10;
2766 assert(IsLog10 || Op.getOpcode() == ISD::FLOG);
2767
2768 if (VT == MVT::f16 || Flags.hasApproximateFuncs()) {
2769 // TODO: The direct f16 path is 1.79 ulp for f16. This should be used
2770 // depending on !fpmath metadata.
2771
2772 bool PromoteToF32 = VT == MVT::f16 && (!Flags.hasApproximateFuncs() ||
2773 !isTypeLegal(MVT::f16));
2774
2775 if (PromoteToF32) {
2776 // Log and multiply in f32 is always good enough for f16.
2777 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2778 }
2779
2780 SDValue Lowered = LowerFLOGUnsafe(X, DL, DAG, IsLog10, Flags);
2781 if (PromoteToF32) {
2782 return DAG.getNode(ISD::FP_ROUND, DL, VT, Lowered,
2783 DAG.getTargetConstant(0, DL, MVT::i32), Flags);
2784 }
2785
2786 return Lowered;
2787 }
2788
2789 SDValue ScaledInput, IsScaled;
2790 if (VT == MVT::f16)
2791 X = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, X, Flags);
2792 else {
2793 std::tie(ScaledInput, IsScaled) = getScaledLogInput(DAG, DL, X, Flags);
2794 if (ScaledInput)
2795 X = ScaledInput;
2796 }
2797
2798 SDValue Y = DAG.getNode(AMDGPUISD::LOG, DL, VT, X, Flags);
2799
2800 SDValue R;
2801 if (Subtarget->hasFastFMAF32()) {
2802 // c+cc are ln(2)/ln(10) to more than 49 bits
2803 const float c_log10 = 0x1.344134p-2f;
2804 const float cc_log10 = 0x1.09f79ep-26f;
2805
2806 // c + cc is ln(2) to more than 49 bits
2807 const float c_log = 0x1.62e42ep-1f;
2808 const float cc_log = 0x1.efa39ep-25f;
2809
2810 SDValue C = DAG.getConstantFP(IsLog10 ? c_log10 : c_log, DL, VT);
2811 SDValue CC = DAG.getConstantFP(IsLog10 ? cc_log10 : cc_log, DL, VT);
2812 // This adds correction terms for which contraction may lead to an increase
2813 // in the error of the approximation, so disable it.
2814 Flags.setAllowContract(false);
2815 R = DAG.getNode(ISD::FMUL, DL, VT, Y, C, Flags);
2816 SDValue NegR = DAG.getNode(ISD::FNEG, DL, VT, R, Flags);
2817 SDValue FMA0 = DAG.getNode(ISD::FMA, DL, VT, Y, C, NegR, Flags);
2818 SDValue FMA1 = DAG.getNode(ISD::FMA, DL, VT, Y, CC, FMA0, Flags);
2819 R = DAG.getNode(ISD::FADD, DL, VT, R, FMA1, Flags);
2820 } else {
2821 // ch+ct is ln(2)/ln(10) to more than 36 bits
2822 const float ch_log10 = 0x1.344000p-2f;
2823 const float ct_log10 = 0x1.3509f6p-18f;
2824
2825 // ch + ct is ln(2) to more than 36 bits
2826 const float ch_log = 0x1.62e000p-1f;
2827 const float ct_log = 0x1.0bfbe8p-15f;
2828
2829 SDValue CH = DAG.getConstantFP(IsLog10 ? ch_log10 : ch_log, DL, VT);
2830 SDValue CT = DAG.getConstantFP(IsLog10 ? ct_log10 : ct_log, DL, VT);
2831
2832 SDValue YAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Y);
2833 SDValue MaskConst = DAG.getConstant(0xfffff000, DL, MVT::i32);
2834 SDValue YHInt = DAG.getNode(ISD::AND, DL, MVT::i32, YAsInt, MaskConst);
2835 SDValue YH = DAG.getNode(ISD::BITCAST, DL, MVT::f32, YHInt);
2836 SDValue YT = DAG.getNode(ISD::FSUB, DL, VT, Y, YH, Flags);
2837 // This adds correction terms for which contraction may lead to an increase
2838 // in the error of the approximation, so disable it.
2839 Flags.setAllowContract(false);
2840 SDValue YTCT = DAG.getNode(ISD::FMUL, DL, VT, YT, CT, Flags);
2841 SDValue Mad0 = getMad(DAG, DL, VT, YH, CT, YTCT, Flags);
2842 SDValue Mad1 = getMad(DAG, DL, VT, YT, CH, Mad0, Flags);
2843 R = getMad(DAG, DL, VT, YH, CH, Mad1);
2844 }
2845
2846 const bool IsFiniteOnly = Flags.hasNoNaNs() && Flags.hasNoInfs();
2847
2848 // TODO: Check if known finite from source value.
2849 if (!IsFiniteOnly) {
2850 SDValue IsFinite = getIsFinite(DAG, Y, Flags);
2851 R = DAG.getNode(ISD::SELECT, DL, VT, IsFinite, R, Y, Flags);
2852 }
2853
2854 if (IsScaled) {
2855 SDValue Zero = DAG.getConstantFP(0.0f, DL, VT);
2856 SDValue ShiftK =
2857 DAG.getConstantFP(IsLog10 ? 0x1.344136p+3f : 0x1.62e430p+4f, DL, VT);
2858 SDValue Shift =
2859 DAG.getNode(ISD::SELECT, DL, VT, IsScaled, ShiftK, Zero, Flags);
2860 R = DAG.getNode(ISD::FSUB, DL, VT, R, Shift, Flags);
2861 }
2862
2863 return R;
2864}
2865
2869
2870// Do f32 fast math expansion for flog2 or flog10. This is accurate enough for a
2871// promote f16 operation.
2873 SelectionDAG &DAG, bool IsLog10,
2874 SDNodeFlags Flags) const {
2875 EVT VT = Src.getValueType();
2876 unsigned LogOp =
2877 VT == MVT::f32 ? (unsigned)AMDGPUISD::LOG : (unsigned)ISD::FLOG2;
2878
2879 double Log2BaseInverted =
2881
2882 if (VT == MVT::f32) {
2883 auto [ScaledInput, IsScaled] = getScaledLogInput(DAG, SL, Src, Flags);
2884 if (ScaledInput) {
2885 SDValue LogSrc = DAG.getNode(AMDGPUISD::LOG, SL, VT, ScaledInput, Flags);
2886 SDValue ScaledResultOffset =
2887 DAG.getConstantFP(-32.0 * Log2BaseInverted, SL, VT);
2888
2889 SDValue Zero = DAG.getConstantFP(0.0f, SL, VT);
2890
2891 SDValue ResultOffset = DAG.getNode(ISD::SELECT, SL, VT, IsScaled,
2892 ScaledResultOffset, Zero, Flags);
2893
2894 SDValue Log2Inv = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2895
2896 if (Subtarget->hasFastFMAF32())
2897 return DAG.getNode(ISD::FMA, SL, VT, LogSrc, Log2Inv, ResultOffset,
2898 Flags);
2899 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, LogSrc, Log2Inv, Flags);
2900 return DAG.getNode(ISD::FADD, SL, VT, Mul, ResultOffset);
2901 }
2902 }
2903
2904 SDValue Log2Operand = DAG.getNode(LogOp, SL, VT, Src, Flags);
2905 SDValue Log2BaseInvertedOperand = DAG.getConstantFP(Log2BaseInverted, SL, VT);
2906
2907 return DAG.getNode(ISD::FMUL, SL, VT, Log2Operand, Log2BaseInvertedOperand,
2908 Flags);
2909}
2910
2911// This expansion gives a result slightly better than 1ulp.
2913 SelectionDAG &DAG) const {
2914 SDLoc DL(Op);
2915 SDValue X = Op.getOperand(0);
2916
2917 // TODO: Check if reassoc is safe. There is an output change in exp2 and
2918 // exp10, which slightly increases ulp.
2919 SDNodeFlags Flags = Op->getFlags() & ~SDNodeFlags::AllowReassociation;
2920
2921 SDValue DN, F, T;
2922
2923 if (Op.getOpcode() == ISD::FEXP2) {
2924 // dn = rint(x)
2925 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, X, Flags);
2926 // f = x - dn
2927 F = DAG.getNode(ISD::FSUB, DL, MVT::f64, X, DN, Flags);
2928 // t = f*C1 + f*C2
2929 SDValue C1 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2930 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2931 SDValue Mul2 = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C2, Flags);
2932 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C1, Mul2, Flags);
2933 } else if (Op.getOpcode() == ISD::FEXP10) {
2934 // dn = rint(x * C1)
2935 SDValue C1 = DAG.getConstantFP(0x1.a934f0979a371p+1, DL, MVT::f64);
2936 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2937 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2938
2939 // f = FMA(-dn, C2, FMA(-dn, C3, x))
2940 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2941 SDValue C2 = DAG.getConstantFP(-0x1.9dc1da994fd21p-59, DL, MVT::f64);
2942 SDValue C3 = DAG.getConstantFP(0x1.34413509f79ffp-2, DL, MVT::f64);
2943 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2944 F = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2945
2946 // t = FMA(f, C4, f*C5)
2947 SDValue C4 = DAG.getConstantFP(0x1.26bb1bbb55516p+1, DL, MVT::f64);
2948 SDValue C5 = DAG.getConstantFP(-0x1.f48ad494ea3e9p-53, DL, MVT::f64);
2949 SDValue MulF = DAG.getNode(ISD::FMUL, DL, MVT::f64, F, C5, Flags);
2950 T = DAG.getNode(ISD::FMA, DL, MVT::f64, F, C4, MulF, Flags);
2951 } else { // ISD::FEXP
2952 // dn = rint(x * C1)
2953 SDValue C1 = DAG.getConstantFP(0x1.71547652b82fep+0, DL, MVT::f64);
2954 SDValue Mul = DAG.getNode(ISD::FMUL, DL, MVT::f64, X, C1, Flags);
2955 DN = DAG.getNode(ISD::FRINT, DL, MVT::f64, Mul, Flags);
2956
2957 // t = FMA(-dn, C2, FMA(-dn, C3, x))
2958 SDValue NegDN = DAG.getNode(ISD::FNEG, DL, MVT::f64, DN, Flags);
2959 SDValue C2 = DAG.getConstantFP(0x1.abc9e3b39803fp-56, DL, MVT::f64);
2960 SDValue C3 = DAG.getConstantFP(0x1.62e42fefa39efp-1, DL, MVT::f64);
2961 SDValue Inner = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C3, X, Flags);
2962 T = DAG.getNode(ISD::FMA, DL, MVT::f64, NegDN, C2, Inner, Flags);
2963 }
2964
2965 // Polynomial expansion for p
2966 SDValue P = DAG.getConstantFP(0x1.ade156a5dcb37p-26, DL, MVT::f64);
2967 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2968 DAG.getConstantFP(0x1.28af3fca7ab0cp-22, DL, MVT::f64),
2969 Flags);
2970 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2971 DAG.getConstantFP(0x1.71dee623fde64p-19, DL, MVT::f64),
2972 Flags);
2973 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2974 DAG.getConstantFP(0x1.a01997c89e6b0p-16, DL, MVT::f64),
2975 Flags);
2976 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2977 DAG.getConstantFP(0x1.a01a014761f6ep-13, DL, MVT::f64),
2978 Flags);
2979 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2980 DAG.getConstantFP(0x1.6c16c1852b7b0p-10, DL, MVT::f64),
2981 Flags);
2982 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2983 DAG.getConstantFP(0x1.1111111122322p-7, DL, MVT::f64), Flags);
2984 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2985 DAG.getConstantFP(0x1.55555555502a1p-5, DL, MVT::f64), Flags);
2986 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2987 DAG.getConstantFP(0x1.5555555555511p-3, DL, MVT::f64), Flags);
2988 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P,
2989 DAG.getConstantFP(0x1.000000000000bp-1, DL, MVT::f64), Flags);
2990
2991 SDValue One = DAG.getConstantFP(1.0, DL, MVT::f64);
2992
2993 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
2994 P = DAG.getNode(ISD::FMA, DL, MVT::f64, T, P, One, Flags);
2995
2996 // z = ldexp(p, (int)dn)
2997 SDValue DNInt = DAG.getNode(ISD::FP_TO_SINT, DL, MVT::i32, DN);
2998 SDValue Z = DAG.getNode(ISD::FLDEXP, DL, MVT::f64, P, DNInt, Flags);
2999
3000 // Overflow/underflow guards
3001 SDValue CondHi = DAG.getSetCC(
3002 DL, MVT::i1, X, DAG.getConstantFP(1024.0, DL, MVT::f64), ISD::SETULE);
3003
3004 if (!Flags.hasNoInfs()) {
3005 SDValue PInf = DAG.getConstantFP(std::numeric_limits<double>::infinity(),
3006 DL, MVT::f64);
3007 Z = DAG.getSelect(DL, MVT::f64, CondHi, Z, PInf, Flags);
3008 }
3009
3010 SDValue CondLo = DAG.getSetCC(
3011 DL, MVT::i1, X, DAG.getConstantFP(-1075.0, DL, MVT::f64), ISD::SETUGE);
3012 SDValue Zero = DAG.getConstantFP(0.0, DL, MVT::f64);
3013 Z = DAG.getSelect(DL, MVT::f64, CondLo, Z, Zero, Flags);
3014
3015 return Z;
3016}
3017
3019 // v_exp_f32 is good enough for OpenCL, except it doesn't handle denormals.
3020 // If we have to handle denormals, scale up the input and adjust the result.
3021
3022 EVT VT = Op.getValueType();
3023 if (VT == MVT::f64)
3024 return lowerFEXPF64(Op, DAG);
3025
3026 SDLoc SL(Op);
3027 SDValue Src = Op.getOperand(0);
3028 SDNodeFlags Flags = Op->getFlags();
3029
3030 if (VT == MVT::f16) {
3031 // Nothing in half is a denormal when promoted to f32.
3032 assert(!isTypeLegal(MVT::f16));
3033 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Src, Flags);
3034 SDValue Log = DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Ext, Flags);
3035 return DAG.getNode(ISD::FP_ROUND, SL, VT, Log,
3036 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3037 }
3038
3039 assert(VT == MVT::f32);
3040
3041 if (!needsDenormHandlingF32(DAG, Src, Flags))
3042 return DAG.getNode(AMDGPUISD::EXP, SL, MVT::f32, Src, Flags);
3043
3044 // bool needs_scaling = x < -0x1.f80000p+6f;
3045 // v_exp_f32(x + (s ? 0x1.0p+6f : 0.0f)) * (s ? 0x1.0p-64f : 1.0f);
3046
3047 // -nextafter(128.0, -1)
3048 SDValue RangeCheckConst = DAG.getConstantFP(-0x1.f80000p+6f, SL, VT);
3049
3050 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3051
3052 SDValue NeedsScaling =
3053 DAG.getSetCC(SL, SetCCVT, Src, RangeCheckConst, ISD::SETOLT);
3054
3055 SDValue SixtyFour = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3056 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3057
3058 SDValue AddOffset =
3059 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, SixtyFour, Zero);
3060
3061 SDValue AddInput = DAG.getNode(ISD::FADD, SL, VT, Src, AddOffset, Flags);
3062 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, AddInput, Flags);
3063
3064 SDValue TwoExpNeg64 = DAG.getConstantFP(0x1.0p-64f, SL, VT);
3065 SDValue One = DAG.getConstantFP(1.0, SL, VT);
3066 SDValue ResultScale =
3067 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, TwoExpNeg64, One);
3068
3069 return DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScale, Flags);
3070}
3071
3073 SelectionDAG &DAG,
3074 SDNodeFlags Flags,
3075 bool IsExp10) const {
3076 // exp(x) -> exp2(M_LOG2E_F * x);
3077 // exp10(x) -> exp2(log2(10) * x);
3078 EVT VT = X.getValueType();
3079 SDValue Const =
3080 DAG.getConstantFP(IsExp10 ? 0x1.a934f0p+1f : numbers::log2e, SL, VT);
3081
3082 SDValue Mul = DAG.getNode(ISD::FMUL, SL, VT, X, Const, Flags);
3083 return DAG.getNode(VT == MVT::f32 ? (unsigned)AMDGPUISD::EXP
3084 : (unsigned)ISD::FEXP2,
3085 SL, VT, Mul, Flags);
3086}
3087
3089 SelectionDAG &DAG,
3090 SDNodeFlags Flags) const {
3091 EVT VT = X.getValueType();
3092 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags))
3093 return lowerFEXPUnsafeImpl(X, SL, DAG, Flags, /*IsExp10=*/false);
3094
3095 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3096
3097 SDValue Threshold = DAG.getConstantFP(-0x1.5d58a0p+6f, SL, VT);
3098 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3099
3100 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+6f, SL, VT);
3101
3102 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3103
3104 SDValue AdjustedX =
3105 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3106
3107 const SDValue Log2E = DAG.getConstantFP(numbers::log2e, SL, VT);
3108 SDValue ExpInput = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, Log2E, Flags);
3109
3110 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, ExpInput, Flags);
3111
3112 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.969d48p-93f, SL, VT);
3113 SDValue AdjustedResult =
3114 DAG.getNode(ISD::FMUL, SL, VT, Exp2, ResultScaleFactor, Flags);
3115
3116 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, Exp2,
3117 Flags);
3118}
3119
3120/// Emit approx-funcs appropriate lowering for exp10. inf/nan should still be
3121/// handled correctly.
3123 SelectionDAG &DAG,
3124 SDNodeFlags Flags) const {
3125 const EVT VT = X.getValueType();
3126
3127 const unsigned Exp2Op = VT == MVT::f32 ? static_cast<unsigned>(AMDGPUISD::EXP)
3128 : static_cast<unsigned>(ISD::FEXP2);
3129
3130 if (VT != MVT::f32 || !needsDenormHandlingF32(DAG, X, Flags)) {
3131 // exp2(x * 0x1.a92000p+1f) * exp2(x * 0x1.4f0978p-11f);
3132 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3133 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3134
3135 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, X, K0, Flags);
3136 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3137 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, X, K1, Flags);
3138 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3139 return DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1);
3140 }
3141
3142 // bool s = x < -0x1.2f7030p+5f;
3143 // x += s ? 0x1.0p+5f : 0.0f;
3144 // exp10 = exp2(x * 0x1.a92000p+1f) *
3145 // exp2(x * 0x1.4f0978p-11f) *
3146 // (s ? 0x1.9f623ep-107f : 1.0f);
3147
3148 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3149
3150 SDValue Threshold = DAG.getConstantFP(-0x1.2f7030p+5f, SL, VT);
3151 SDValue NeedsScaling = DAG.getSetCC(SL, SetCCVT, X, Threshold, ISD::SETOLT);
3152
3153 SDValue ScaleOffset = DAG.getConstantFP(0x1.0p+5f, SL, VT);
3154 SDValue ScaledX = DAG.getNode(ISD::FADD, SL, VT, X, ScaleOffset, Flags);
3155 SDValue AdjustedX =
3156 DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, ScaledX, X);
3157
3158 SDValue K0 = DAG.getConstantFP(0x1.a92000p+1f, SL, VT);
3159 SDValue K1 = DAG.getConstantFP(0x1.4f0978p-11f, SL, VT);
3160
3161 SDValue Mul0 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K0, Flags);
3162 SDValue Exp2_0 = DAG.getNode(Exp2Op, SL, VT, Mul0, Flags);
3163 SDValue Mul1 = DAG.getNode(ISD::FMUL, SL, VT, AdjustedX, K1, Flags);
3164 SDValue Exp2_1 = DAG.getNode(Exp2Op, SL, VT, Mul1, Flags);
3165
3166 SDValue MulExps = DAG.getNode(ISD::FMUL, SL, VT, Exp2_0, Exp2_1, Flags);
3167
3168 SDValue ResultScaleFactor = DAG.getConstantFP(0x1.9f623ep-107f, SL, VT);
3169 SDValue AdjustedResult =
3170 DAG.getNode(ISD::FMUL, SL, VT, MulExps, ResultScaleFactor, Flags);
3171
3172 return DAG.getNode(ISD::SELECT, SL, VT, NeedsScaling, AdjustedResult, MulExps,
3173 Flags);
3174}
3175
3177 EVT VT = Op.getValueType();
3178
3179 if (VT == MVT::f64)
3180 return lowerFEXPF64(Op, DAG);
3181
3182 SDLoc SL(Op);
3183 SDValue X = Op.getOperand(0);
3184 SDNodeFlags Flags = Op->getFlags();
3185 const bool IsExp10 = Op.getOpcode() == ISD::FEXP10;
3186
3187 // TODO: Interpret allowApproxFunc as ignoring DAZ. This is currently copying
3188 // library behavior. Also, is known-not-daz source sufficient?
3189 if (allowApproxFunc(DAG, Flags)) { // TODO: Does this really require fast?
3190 return IsExp10 ? lowerFEXP10Unsafe(X, SL, DAG, Flags)
3191 : lowerFEXPUnsafe(X, SL, DAG, Flags);
3192 }
3193
3194 if (VT.getScalarType() == MVT::f16) {
3195 if (VT.isVector())
3196 return SDValue();
3197
3198 // Nothing in half is a denormal when promoted to f32.
3199 //
3200 // exp(f16 x) ->
3201 // fptrunc (v_exp_f32 (fmul (fpext x), log2e))
3202 //
3203 // exp10(f16 x) ->
3204 // fptrunc (v_exp_f32 (fmul (fpext x), log2(10)))
3205 SDValue Ext = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, X, Flags);
3206 SDValue Lowered = lowerFEXPUnsafeImpl(Ext, SL, DAG, Flags, IsExp10);
3207 return DAG.getNode(ISD::FP_ROUND, SL, VT, Lowered,
3208 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
3209 }
3210
3211 assert(VT == MVT::f32);
3212
3213 // Algorithm:
3214 //
3215 // e^x = 2^(x/ln(2)) = 2^(x*(64/ln(2))/64)
3216 //
3217 // x*(64/ln(2)) = n + f, |f| <= 0.5, n is integer
3218 // n = 64*m + j, 0 <= j < 64
3219 //
3220 // e^x = 2^((64*m + j + f)/64)
3221 // = (2^m) * (2^(j/64)) * 2^(f/64)
3222 // = (2^m) * (2^(j/64)) * e^(f*(ln(2)/64))
3223 //
3224 // f = x*(64/ln(2)) - n
3225 // r = f*(ln(2)/64) = x - n*(ln(2)/64)
3226 //
3227 // e^x = (2^m) * (2^(j/64)) * e^r
3228 //
3229 // (2^(j/64)) is precomputed
3230 //
3231 // e^r = 1 + r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3232 // e^r = 1 + q
3233 //
3234 // q = r + (r^2)/2! + (r^3)/3! + (r^4)/4! + (r^5)/5!
3235 //
3236 // e^x = (2^m) * ( (2^(j/64)) + q*(2^(j/64)) )
3237 SDNodeFlags FlagsNoContract = Flags;
3238 FlagsNoContract.setAllowContract(false);
3239
3240 SDValue PH, PL;
3241 if (Subtarget->hasFastFMAF32()) {
3242 const float c_exp = numbers::log2ef;
3243 const float cc_exp = 0x1.4ae0bep-26f; // c+cc are 49 bits
3244 const float c_exp10 = 0x1.a934f0p+1f;
3245 const float cc_exp10 = 0x1.2f346ep-24f;
3246
3247 SDValue C = DAG.getConstantFP(IsExp10 ? c_exp10 : c_exp, SL, VT);
3248 SDValue CC = DAG.getConstantFP(IsExp10 ? cc_exp10 : cc_exp, SL, VT);
3249
3250 PH = DAG.getNode(ISD::FMUL, SL, VT, X, C, Flags);
3251 SDValue NegPH = DAG.getNode(ISD::FNEG, SL, VT, PH, Flags);
3252 SDValue FMA0 = DAG.getNode(ISD::FMA, SL, VT, X, C, NegPH, Flags);
3253 PL = DAG.getNode(ISD::FMA, SL, VT, X, CC, FMA0, Flags);
3254 } else {
3255 const float ch_exp = 0x1.714000p+0f;
3256 const float cl_exp = 0x1.47652ap-12f; // ch + cl are 36 bits
3257
3258 const float ch_exp10 = 0x1.a92000p+1f;
3259 const float cl_exp10 = 0x1.4f0978p-11f;
3260
3261 SDValue CH = DAG.getConstantFP(IsExp10 ? ch_exp10 : ch_exp, SL, VT);
3262 SDValue CL = DAG.getConstantFP(IsExp10 ? cl_exp10 : cl_exp, SL, VT);
3263
3264 SDValue XAsInt = DAG.getNode(ISD::BITCAST, SL, MVT::i32, X);
3265 SDValue MaskConst = DAG.getConstant(0xfffff000, SL, MVT::i32);
3266 SDValue XHAsInt = DAG.getNode(ISD::AND, SL, MVT::i32, XAsInt, MaskConst);
3267 SDValue XH = DAG.getNode(ISD::BITCAST, SL, VT, XHAsInt);
3268 SDValue XL = DAG.getNode(ISD::FSUB, SL, VT, X, XH, Flags);
3269
3270 PH = DAG.getNode(ISD::FMUL, SL, VT, XH, CH, Flags);
3271
3272 SDValue XLCL = DAG.getNode(ISD::FMUL, SL, VT, XL, CL, Flags);
3273 SDValue Mad0 = getMad(DAG, SL, VT, XL, CH, XLCL, Flags);
3274 PL = getMad(DAG, SL, VT, XH, CL, Mad0, Flags);
3275 }
3276
3277 SDValue E = DAG.getNode(ISD::FROUNDEVEN, SL, VT, PH, Flags);
3278
3279 // It is unsafe to contract this fsub into the PH multiply.
3280 SDValue PHSubE = DAG.getNode(ISD::FSUB, SL, VT, PH, E, FlagsNoContract);
3281
3282 SDValue A = DAG.getNode(ISD::FADD, SL, VT, PHSubE, PL, Flags);
3283 SDValue IntE = DAG.getNode(ISD::FP_TO_SINT, SL, MVT::i32, E);
3284 SDValue Exp2 = DAG.getNode(AMDGPUISD::EXP, SL, VT, A, Flags);
3285
3286 SDValue R = DAG.getNode(ISD::FLDEXP, SL, VT, Exp2, IntE, Flags);
3287
3288 SDValue UnderflowCheckConst =
3289 DAG.getConstantFP(IsExp10 ? -0x1.66d3e8p+5f : -0x1.9d1da0p+6f, SL, VT);
3290
3291 EVT SetCCVT = getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), VT);
3292 SDValue Zero = DAG.getConstantFP(0.0, SL, VT);
3293 SDValue Underflow =
3294 DAG.getSetCC(SL, SetCCVT, X, UnderflowCheckConst, ISD::SETOLT);
3295
3296 R = DAG.getNode(ISD::SELECT, SL, VT, Underflow, Zero, R);
3297
3298 if (!Flags.hasNoInfs()) {
3299 SDValue OverflowCheckConst =
3300 DAG.getConstantFP(IsExp10 ? 0x1.344136p+5f : 0x1.62e430p+6f, SL, VT);
3301 SDValue Overflow =
3302 DAG.getSetCC(SL, SetCCVT, X, OverflowCheckConst, ISD::SETOGT);
3303 SDValue Inf =
3305 R = DAG.getNode(ISD::SELECT, SL, VT, Overflow, Inf, R);
3306 }
3307
3308 return R;
3309}
3310
3311static bool isCtlzOpc(unsigned Opc) {
3312 return Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON;
3313}
3314
3315static bool isCttzOpc(unsigned Opc) {
3316 return Opc == ISD::CTTZ || Opc == ISD::CTTZ_ZERO_POISON;
3317}
3318
3320 SelectionDAG &DAG) const {
3321 auto SL = SDLoc(Op);
3322 auto Opc = Op.getOpcode();
3323 auto Arg = Op.getOperand(0u);
3324 auto ResultVT = Op.getValueType();
3325
3326 if (ResultVT != MVT::i8 && ResultVT != MVT::i16)
3327 return {};
3328
3330 assert(ResultVT == Arg.getValueType());
3331
3332 const uint64_t NumBits = ResultVT.getFixedSizeInBits();
3333 SDValue NumExtBits = DAG.getConstant(32u - NumBits, SL, MVT::i32);
3334 SDValue NewOp;
3335
3336 if (Opc == ISD::CTLZ_ZERO_POISON) {
3337 NewOp = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Arg);
3338 NewOp = DAG.getNode(ISD::SHL, SL, MVT::i32, NewOp, NumExtBits);
3339 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3340 } else {
3341 NewOp = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Arg);
3342 NewOp = DAG.getNode(Opc, SL, MVT::i32, NewOp);
3343 NewOp = DAG.getNode(ISD::SUB, SL, MVT::i32, NewOp, NumExtBits);
3344 }
3345
3346 return DAG.getNode(ISD::TRUNCATE, SL, ResultVT, NewOp);
3347}
3348
3350 SDLoc SL(Op);
3351 SDValue Src = Op.getOperand(0);
3352
3353 assert(isCtlzOpc(Op.getOpcode()) || isCttzOpc(Op.getOpcode()));
3354 bool Ctlz = isCtlzOpc(Op.getOpcode());
3355 unsigned NewOpc = Ctlz ? AMDGPUISD::FFBH_U32 : AMDGPUISD::FFBL_B32;
3356
3357 bool ZeroUndef = Op.getOpcode() == ISD::CTLZ_ZERO_POISON ||
3358 Op.getOpcode() == ISD::CTTZ_ZERO_POISON;
3359 bool Is64BitScalar = !Src->isDivergent() && Src.getValueType() == MVT::i64;
3360
3361 if (Src.getValueType() == MVT::i32 || Is64BitScalar) {
3362 // (ctlz hi:lo) -> (umin (ffbh src), 32)
3363 // (cttz hi:lo) -> (umin (ffbl src), 32)
3364 // (ctlz_zero_poison src) -> (ffbh src)
3365 // (cttz_zero_poison src) -> (ffbl src)
3366
3367 // 64-bit scalar version produce 32-bit result
3368 // (ctlz hi:lo) -> (umin (S_FLBIT_I32_B64 src), 64)
3369 // (cttz hi:lo) -> (umin (S_FF1_I32_B64 src), 64)
3370 // (ctlz_zero_poison src) -> (S_FLBIT_I32_B64 src)
3371 // (cttz_zero_poison src) -> (S_FF1_I32_B64 src)
3372 SDValue NewOpr = DAG.getNode(NewOpc, SL, MVT::i32, Src);
3373 if (!ZeroUndef) {
3374 const SDValue ConstVal = DAG.getConstant(
3375 Op.getValueType().getScalarSizeInBits(), SL, MVT::i32);
3376 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, ConstVal);
3377 }
3378 return DAG.getNode(ISD::ZERO_EXTEND, SL, Src.getValueType(), NewOpr);
3379 }
3380
3381 SDValue Lo, Hi;
3382 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3383
3384 SDValue OprLo = DAG.getNode(NewOpc, SL, MVT::i32, Lo);
3385 SDValue OprHi = DAG.getNode(NewOpc, SL, MVT::i32, Hi);
3386
3387 // (ctlz hi:lo) -> (umin3 (ffbh hi), (uaddsat (ffbh lo), 32), 64)
3388 // (cttz hi:lo) -> (umin3 (uaddsat (ffbl hi), 32), (ffbl lo), 64)
3389 // (ctlz_zero_poison hi:lo) -> (umin (ffbh hi), (add (ffbh lo), 32))
3390 // (cttz_zero_poison hi:lo) -> (umin (add (ffbl hi), 32), (ffbl lo))
3391
3392 unsigned AddOpc = ZeroUndef ? ISD::ADD : ISD::UADDSAT;
3393 const SDValue Const32 = DAG.getConstant(32, SL, MVT::i32);
3394 if (Ctlz)
3395 OprLo = DAG.getNode(AddOpc, SL, MVT::i32, OprLo, Const32);
3396 else
3397 OprHi = DAG.getNode(AddOpc, SL, MVT::i32, OprHi, Const32);
3398
3399 SDValue NewOpr;
3400 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, OprLo, OprHi);
3401 if (!ZeroUndef) {
3402 const SDValue Const64 = DAG.getConstant(64, SL, MVT::i32);
3403 NewOpr = DAG.getNode(ISD::UMIN, SL, MVT::i32, NewOpr, Const64);
3404 }
3405
3406 return DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i64, NewOpr);
3407}
3408
3410 SDLoc SL(Op);
3411 SDValue Src = Op.getOperand(0);
3412 assert(Src.getValueType() == MVT::i32 && "LowerCTLS only supports i32");
3413 SDValue Ffbh = DAG.getNode(
3414 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3415 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Src);
3416 SDValue Clamped = DAG.getNode(ISD::UMIN, SL, MVT::i32, Ffbh,
3417 DAG.getConstant(32, SL, MVT::i32));
3418 return DAG.getNode(ISD::ADD, SL, MVT::i32, Clamped,
3419 DAG.getAllOnesConstant(SL, MVT::i32));
3420}
3421
3423 EVT FP16Ty) const {
3424 assert(FP16Ty == MVT::f16 || FP16Ty == MVT::bf16);
3425 SDLoc SL(Op);
3426 SDValue Src = Op.getOperand(0);
3427 SDValue ToF32 = DAG.getNode(Op.getOpcode(), SL, MVT::f32, Src);
3428 SDValue FPRoundFlag = DAG.getIntPtrConstant(0, SL, /*isTarget=*/true);
3429 return DAG.getNode(ISD::FP_ROUND, SL, FP16Ty, ToF32, FPRoundFlag);
3430}
3431
3433 bool Signed) const {
3434 // The regular method converting a 64-bit integer to float roughly consists of
3435 // 2 steps: normalization and rounding. In fact, after normalization, the
3436 // conversion from a 64-bit integer to a float is essentially the same as the
3437 // one from a 32-bit integer. The only difference is that it has more
3438 // trailing bits to be rounded. To leverage the native 32-bit conversion, a
3439 // 64-bit integer could be preprocessed and fit into a 32-bit integer then
3440 // converted into the correct float number. The basic steps for the unsigned
3441 // conversion are illustrated in the following pseudo code:
3442 //
3443 // f32 uitofp(i64 u) {
3444 // i32 hi, lo = split(u);
3445 // // Only count the leading zeros in hi as we have native support of the
3446 // // conversion from i32 to f32. If hi is all 0s, the conversion is
3447 // // reduced to a 32-bit one automatically.
3448 // i32 shamt = clz(hi); // Return 32 if hi is all 0s.
3449 // u <<= shamt;
3450 // hi, lo = split(u);
3451 // hi |= (lo != 0) ? 1 : 0; // Adjust rounding bit in hi based on lo.
3452 // // convert it as a 32-bit integer and scale the result back.
3453 // return uitofp(hi) * 2^(32 - shamt);
3454 // }
3455 //
3456 // The signed one follows the same principle but uses 'ffbh_i32' to count its
3457 // sign bits instead. If 'ffbh_i32' is not available, its absolute value is
3458 // converted instead followed by negation based its sign bit.
3459
3460 SDLoc SL(Op);
3461 SDValue Src = Op.getOperand(0);
3462
3463 SDValue Lo, Hi;
3464 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3465 SDValue Sign;
3466 SDValue ShAmt;
3467 if (Signed && Subtarget->isGCN()) {
3468 // We also need to consider the sign bit in Lo if Hi has just sign bits,
3469 // i.e. Hi is 0 or -1. However, that only needs to take the MSB into
3470 // account. That is, the maximal shift is
3471 // - 32 if Lo and Hi have opposite signs;
3472 // - 33 if Lo and Hi have the same sign.
3473 //
3474 // Or, MaxShAmt = 33 + OppositeSign, where
3475 //
3476 // OppositeSign is defined as ((Lo ^ Hi) >> 31), which is
3477 // - -1 if Lo and Hi have opposite signs; and
3478 // - 0 otherwise.
3479 //
3480 // All in all, ShAmt is calculated as
3481 //
3482 // umin(sffbh(Hi), 33 + (Lo^Hi)>>31) - 1.
3483 //
3484 // or
3485 //
3486 // umin(sffbh(Hi) - 1, 32 + (Lo^Hi)>>31).
3487 //
3488 // to reduce the critical path.
3489 SDValue OppositeSign = DAG.getNode(
3490 ISD::SRA, SL, MVT::i32, DAG.getNode(ISD::XOR, SL, MVT::i32, Lo, Hi),
3491 DAG.getConstant(31, SL, MVT::i32));
3492 SDValue MaxShAmt =
3493 DAG.getNode(ISD::ADD, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3494 OppositeSign);
3495 // Count the leading sign bits.
3496 ShAmt = DAG.getNode(
3497 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
3498 DAG.getTargetConstant(Intrinsic::amdgcn_sffbh, SL, MVT::i32), Hi);
3499 // Different from unsigned conversion, the shift should be one bit less to
3500 // preserve the sign bit.
3501 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, ShAmt,
3502 DAG.getConstant(1, SL, MVT::i32));
3503 ShAmt = DAG.getNode(ISD::UMIN, SL, MVT::i32, ShAmt, MaxShAmt);
3504 } else {
3505 if (Signed) {
3506 // Without 'ffbh_i32', only leading zeros could be counted. Take the
3507 // absolute value first.
3508 Sign = DAG.getNode(ISD::SRA, SL, MVT::i64, Src,
3509 DAG.getConstant(63, SL, MVT::i64));
3510 SDValue Abs =
3511 DAG.getNode(ISD::XOR, SL, MVT::i64,
3512 DAG.getNode(ISD::ADD, SL, MVT::i64, Src, Sign), Sign);
3513 std::tie(Lo, Hi) = split64BitValue(Abs, DAG);
3514 }
3515 // Count the leading zeros.
3516 ShAmt = DAG.getNode(ISD::CTLZ, SL, MVT::i32, Hi);
3517 // The shift amount for signed integers is [0, 32].
3518 }
3519 // Normalize the given 64-bit integer.
3520 SDValue Norm = DAG.getNode(ISD::SHL, SL, MVT::i64, Src, ShAmt);
3521 // Split it again.
3522 std::tie(Lo, Hi) = split64BitValue(Norm, DAG);
3523 // Calculate the adjust bit for rounding.
3524 // (lo != 0) ? 1 : 0 => (lo >= 1) ? 1 : 0 => umin(1, lo)
3525 SDValue Adjust = DAG.getNode(ISD::UMIN, SL, MVT::i32,
3526 DAG.getConstant(1, SL, MVT::i32), Lo);
3527 // Get the 32-bit normalized integer.
3528 Norm = DAG.getNode(ISD::OR, SL, MVT::i32, Hi, Adjust);
3529 // Convert the normalized 32-bit integer into f32.
3530
3531 bool UseLDEXP = isOperationLegal(ISD::FLDEXP, MVT::f32);
3532 unsigned Opc = Signed && UseLDEXP ? ISD::SINT_TO_FP : ISD::UINT_TO_FP;
3533 SDValue FVal = DAG.getNode(Opc, SL, MVT::f32, Norm);
3534
3535 // Finally, need to scale back the converted floating number as the original
3536 // 64-bit integer is converted as a 32-bit one.
3537 ShAmt = DAG.getNode(ISD::SUB, SL, MVT::i32, DAG.getConstant(32, SL, MVT::i32),
3538 ShAmt);
3539 // On GCN, use LDEXP directly.
3540 if (UseLDEXP)
3541 return DAG.getNode(ISD::FLDEXP, SL, MVT::f32, FVal, ShAmt);
3542
3543 // Otherwise, align 'ShAmt' to the exponent part and add it into the exponent
3544 // part directly to emulate the multiplication of 2^ShAmt. That 8-bit
3545 // exponent is enough to avoid overflowing into the sign bit.
3546 SDValue Exp = DAG.getNode(ISD::SHL, SL, MVT::i32, ShAmt,
3547 DAG.getConstant(23, SL, MVT::i32));
3548 SDValue IVal =
3549 DAG.getNode(ISD::ADD, SL, MVT::i32,
3550 DAG.getNode(ISD::BITCAST, SL, MVT::i32, FVal), Exp);
3551 if (Signed) {
3552 // Set the sign bit.
3553 Sign = DAG.getNode(ISD::SHL, SL, MVT::i32,
3554 DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Sign),
3555 DAG.getConstant(31, SL, MVT::i32));
3556 IVal = DAG.getNode(ISD::OR, SL, MVT::i32, IVal, Sign);
3557 }
3558 return DAG.getNode(ISD::BITCAST, SL, MVT::f32, IVal);
3559}
3560
3562 bool Signed) const {
3563 SDLoc SL(Op);
3564 SDValue Src = Op.getOperand(0);
3565
3566 SDValue Lo, Hi;
3567 std::tie(Lo, Hi) = split64BitValue(Src, DAG);
3568
3570 SL, MVT::f64, Hi);
3571
3572 SDValue CvtLo = DAG.getNode(ISD::UINT_TO_FP, SL, MVT::f64, Lo);
3573
3574 SDValue LdExp = DAG.getNode(ISD::FLDEXP, SL, MVT::f64, CvtHi,
3575 DAG.getConstant(32, SL, MVT::i32));
3576 // TODO: Should this propagate fast-math-flags?
3577 return DAG.getNode(ISD::FADD, SL, MVT::f64, LdExp, CvtLo);
3578}
3579
3581 SelectionDAG &DAG) const {
3582 // TODO: Factor out code common with LowerSINT_TO_FP.
3583 EVT DestVT = Op.getValueType();
3584 SDValue Src = Op.getOperand(0);
3585 EVT SrcVT = Src.getValueType();
3586
3587 if (SrcVT == MVT::i16) {
3588 if (DestVT == MVT::f16)
3589 return Op;
3590 SDLoc DL(Op);
3591
3592 // Promote src to i32
3593 SDValue Ext = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, Src);
3594 return DAG.getNode(ISD::UINT_TO_FP, DL, DestVT, Ext);
3595 }
3596
3597 if (DestVT == MVT::bf16 || DestVT == MVT::f16)
3598 return LowerINT_TO_FP16(Op, DAG, DestVT);
3599
3600 if (SrcVT != MVT::i64)
3601 return Op;
3602
3603 if (DestVT == MVT::f32)
3604 return LowerINT_TO_FP32(Op, DAG, false);
3605
3606 assert(DestVT == MVT::f64);
3607 return LowerINT_TO_FP64(Op, DAG, false);
3608}
3609
3611 SelectionDAG &DAG) const {
3612 EVT DestVT = Op.getValueType();
3613
3614 SDValue Src = Op.getOperand(0);
3615 EVT SrcVT = Src.getValueType();
3616
3617 if (SrcVT == MVT::i16) {
3618 if (DestVT == MVT::f16)
3619 return Op;
3620
3621 SDLoc DL(Op);
3622 // Promote src to i32
3623 SDValue Ext = DAG.getNode(ISD::SIGN_EXTEND, DL, MVT::i32, Src);
3624 return DAG.getNode(ISD::SINT_TO_FP, DL, DestVT, Ext);
3625 }
3626
3627 if (DestVT == MVT::bf16 || DestVT == MVT::f16)
3628 return LowerINT_TO_FP16(Op, DAG, DestVT);
3629
3630 if (SrcVT != MVT::i64)
3631 return Op;
3632
3633 // TODO: Factor out code common with LowerUINT_TO_FP.
3634
3635 if (DestVT == MVT::f32)
3636 return LowerINT_TO_FP32(Op, DAG, true);
3637
3638 assert(DestVT == MVT::f64);
3639 return LowerINT_TO_FP64(Op, DAG, true);
3640}
3641
3643 bool Signed) const {
3644 SDLoc SL(Op);
3645
3646 SDValue Src = Op.getOperand(0);
3647 EVT SrcVT = Src.getValueType();
3648
3649 assert(SrcVT == MVT::f32 || SrcVT == MVT::f64);
3650
3651 // The basic idea of converting a floating point number into a pair of 32-bit
3652 // integers is illustrated as follows:
3653 //
3654 // tf := trunc(val);
3655 // hif := floor(tf * 2^-32);
3656 // lof := tf - hif * 2^32; // lof is always positive due to floor.
3657 // hi := fptoi(hif);
3658 // lo := fptoi(lof);
3659 //
3660 SDValue Trunc = DAG.getNode(ISD::FTRUNC, SL, SrcVT, Src);
3661 SDValue Sign;
3662 if (Signed && SrcVT == MVT::f32) {
3663 // However, a 32-bit floating point number has only 23 bits mantissa and
3664 // it's not enough to hold all the significant bits of `lof` if val is
3665 // negative. To avoid the loss of precision, We need to take the absolute
3666 // value after truncating and flip the result back based on the original
3667 // signedness.
3668 Sign = DAG.getNode(ISD::SRA, SL, MVT::i32,
3669 DAG.getNode(ISD::BITCAST, SL, MVT::i32, Trunc),
3670 DAG.getConstant(31, SL, MVT::i32));
3671 Trunc = DAG.getNode(ISD::FABS, SL, SrcVT, Trunc);
3672 }
3673
3674 SDValue K0, K1;
3675 if (SrcVT == MVT::f64) {
3676 K0 = DAG.getConstantFP(
3677 llvm::bit_cast<double>(UINT64_C(/*2^-32*/ 0x3df0000000000000)), SL,
3678 SrcVT);
3679 K1 = DAG.getConstantFP(
3680 llvm::bit_cast<double>(UINT64_C(/*-2^32*/ 0xc1f0000000000000)), SL,
3681 SrcVT);
3682 } else {
3683 K0 = DAG.getConstantFP(
3684 llvm::bit_cast<float>(UINT32_C(/*2^-32*/ 0x2f800000)), SL, SrcVT);
3685 K1 = DAG.getConstantFP(
3686 llvm::bit_cast<float>(UINT32_C(/*-2^32*/ 0xcf800000)), SL, SrcVT);
3687 }
3688 // TODO: Should this propagate fast-math-flags?
3689 SDValue Mul = DAG.getNode(ISD::FMUL, SL, SrcVT, Trunc, K0);
3690
3691 SDValue FloorMul = DAG.getNode(ISD::FFLOOR, SL, SrcVT, Mul);
3692
3693 SDValue Fma = DAG.getNode(ISD::FMA, SL, SrcVT, FloorMul, K1, Trunc);
3694
3695 SDValue Hi = DAG.getNode((Signed && SrcVT == MVT::f64) ? ISD::FP_TO_SINT
3697 SL, MVT::i32, FloorMul);
3698 SDValue Lo = DAG.getNode(ISD::FP_TO_UINT, SL, MVT::i32, Fma);
3699
3700 SDValue Result = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3701 DAG.getBuildVector(MVT::v2i32, SL, {Lo, Hi}));
3702
3703 if (Signed && SrcVT == MVT::f32) {
3704 assert(Sign);
3705 // Flip the result based on the signedness, which is either all 0s or 1s.
3706 Sign = DAG.getNode(ISD::BITCAST, SL, MVT::i64,
3707 DAG.getBuildVector(MVT::v2i32, SL, {Sign, Sign}));
3708 // r := xor(r, sign) - sign;
3709 Result =
3710 DAG.getNode(ISD::SUB, SL, MVT::i64,
3711 DAG.getNode(ISD::XOR, SL, MVT::i64, Result, Sign), Sign);
3712 }
3713
3714 return Result;
3715}
3716
3718 SDLoc DL(Op);
3719 SDValue N0 = Op.getOperand(0);
3720
3721 // Convert to target node to get known bits
3722 if (N0.getValueType() == MVT::f32)
3723 return DAG.getNode(AMDGPUISD::FP_TO_FP16, DL, Op.getValueType(), N0);
3724
3725 if (Op->getFlags().hasApproximateFuncs()) {
3726 // There is a generic expand for FP_TO_FP16 with unsafe fast math.
3727 return SDValue();
3728 }
3729
3730 return LowerF64ToF16Safe(N0, DL, DAG);
3731}
3732
3733// return node in i32
3735 SelectionDAG &DAG) const {
3736 assert(Src.getSimpleValueType() == MVT::f64);
3737
3738 // f64 -> f16 conversion using round-to-nearest-even rounding mode.
3739 // TODO: We can generate better code for True16.
3740 const unsigned ExpMask = 0x7ff;
3741 const unsigned ExpBiasf64 = 1023;
3742 const unsigned ExpBiasf16 = 15;
3743 SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
3744 SDValue One = DAG.getConstant(1, DL, MVT::i32);
3745 SDValue U = DAG.getNode(ISD::BITCAST, DL, MVT::i64, Src);
3746 SDValue UH = DAG.getNode(ISD::SRL, DL, MVT::i64, U,
3747 DAG.getConstant(32, DL, MVT::i64));
3748 UH = DAG.getZExtOrTrunc(UH, DL, MVT::i32);
3749 U = DAG.getZExtOrTrunc(U, DL, MVT::i32);
3750 SDValue E = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3751 DAG.getConstant(20, DL, MVT::i64));
3752 E = DAG.getNode(ISD::AND, DL, MVT::i32, E,
3753 DAG.getConstant(ExpMask, DL, MVT::i32));
3754 // Subtract the fp64 exponent bias (1023) to get the real exponent and
3755 // add the f16 bias (15) to get the biased exponent for the f16 format.
3756 E = DAG.getNode(ISD::ADD, DL, MVT::i32, E,
3757 DAG.getConstant(-ExpBiasf64 + ExpBiasf16, DL, MVT::i32));
3758
3759 SDValue M = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3760 DAG.getConstant(8, DL, MVT::i32));
3761 M = DAG.getNode(ISD::AND, DL, MVT::i32, M,
3762 DAG.getConstant(0xffe, DL, MVT::i32));
3763
3764 SDValue MaskedSig = DAG.getNode(ISD::AND, DL, MVT::i32, UH,
3765 DAG.getConstant(0x1ff, DL, MVT::i32));
3766 MaskedSig = DAG.getNode(ISD::OR, DL, MVT::i32, MaskedSig, U);
3767
3768 SDValue Lo40Set = DAG.getSelectCC(DL, MaskedSig, Zero, Zero, One, ISD::SETEQ);
3769 M = DAG.getNode(ISD::OR, DL, MVT::i32, M, Lo40Set);
3770
3771 // (M != 0 ? 0x0200 : 0) | 0x7c00;
3772 SDValue I = DAG.getNode(ISD::OR, DL, MVT::i32,
3773 DAG.getSelectCC(DL, M, Zero, DAG.getConstant(0x0200, DL, MVT::i32),
3774 Zero, ISD::SETNE), DAG.getConstant(0x7c00, DL, MVT::i32));
3775
3776 // N = M | (E << 12);
3777 SDValue N = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3778 DAG.getNode(ISD::SHL, DL, MVT::i32, E,
3779 DAG.getConstant(12, DL, MVT::i32)));
3780
3781 // B = clamp(1-E, 0, 13);
3782 SDValue OneSubExp = DAG.getNode(ISD::SUB, DL, MVT::i32,
3783 One, E);
3784 SDValue B = DAG.getNode(ISD::SMAX, DL, MVT::i32, OneSubExp, Zero);
3785 B = DAG.getNode(ISD::SMIN, DL, MVT::i32, B,
3786 DAG.getConstant(13, DL, MVT::i32));
3787
3788 SDValue SigSetHigh = DAG.getNode(ISD::OR, DL, MVT::i32, M,
3789 DAG.getConstant(0x1000, DL, MVT::i32));
3790
3791 SDValue D = DAG.getNode(ISD::SRL, DL, MVT::i32, SigSetHigh, B);
3792 SDValue D0 = DAG.getNode(ISD::SHL, DL, MVT::i32, D, B);
3793 SDValue D1 = DAG.getSelectCC(DL, D0, SigSetHigh, One, Zero, ISD::SETNE);
3794 D = DAG.getNode(ISD::OR, DL, MVT::i32, D, D1);
3795
3796 SDValue V = DAG.getSelectCC(DL, E, One, D, N, ISD::SETLT);
3797 SDValue VLow3 = DAG.getNode(ISD::AND, DL, MVT::i32, V,
3798 DAG.getConstant(0x7, DL, MVT::i32));
3799 V = DAG.getNode(ISD::SRL, DL, MVT::i32, V,
3800 DAG.getConstant(2, DL, MVT::i32));
3801 SDValue V0 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(3, DL, MVT::i32),
3802 One, Zero, ISD::SETEQ);
3803 SDValue V1 = DAG.getSelectCC(DL, VLow3, DAG.getConstant(5, DL, MVT::i32),
3804 One, Zero, ISD::SETGT);
3805 V1 = DAG.getNode(ISD::OR, DL, MVT::i32, V0, V1);
3806 V = DAG.getNode(ISD::ADD, DL, MVT::i32, V, V1);
3807
3808 V = DAG.getSelectCC(DL, E, DAG.getConstant(30, DL, MVT::i32),
3809 DAG.getConstant(0x7c00, DL, MVT::i32), V, ISD::SETGT);
3810 V = DAG.getSelectCC(DL, E, DAG.getConstant(1039, DL, MVT::i32),
3811 I, V, ISD::SETEQ);
3812
3813 // Extract the sign bit.
3814 SDValue Sign = DAG.getNode(ISD::SRL, DL, MVT::i32, UH,
3815 DAG.getConstant(16, DL, MVT::i32));
3816 Sign = DAG.getNode(ISD::AND, DL, MVT::i32, Sign,
3817 DAG.getConstant(0x8000, DL, MVT::i32));
3818
3819 return DAG.getNode(ISD::OR, DL, MVT::i32, Sign, V);
3820}
3821
3823 SelectionDAG &DAG) const {
3824 SDValue Src = Op.getOperand(0);
3825 unsigned OpOpcode = Op.getOpcode();
3826 EVT SrcVT = Src.getValueType();
3827 EVT DestVT = Op.getValueType();
3828
3829 // Will be selected natively
3830 if (SrcVT == MVT::f16 && DestVT == MVT::i16)
3831 return Op;
3832
3833 if (SrcVT == MVT::bf16 || (SrcVT == MVT::f16 && DestVT == MVT::i32)) {
3834 SDLoc DL(Op);
3835 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
3836 return DAG.getNode(Op.getOpcode(), DL, DestVT, PromotedSrc);
3837 }
3838
3839 // Promote i16 to i32
3840 if (DestVT == MVT::i16 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) {
3841 SDLoc DL(Op);
3842
3843 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3844 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToInt32);
3845 }
3846
3847 if (DestVT != MVT::i64)
3848 return Op;
3849
3850 if (SrcVT == MVT::f16 ||
3851 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP)) {
3852 SDLoc DL(Op);
3853
3854 SDValue FpToInt32 = DAG.getNode(OpOpcode, DL, MVT::i32, Src);
3855 unsigned Ext =
3857 return DAG.getNode(Ext, DL, MVT::i64, FpToInt32);
3858 }
3859
3860 if (SrcVT == MVT::f32 || SrcVT == MVT::f64)
3861 return LowerFP_TO_INT64(Op, DAG, OpOpcode == ISD::FP_TO_SINT);
3862
3863 return SDValue();
3864}
3865
3867 SelectionDAG &DAG) const {
3868 SDValue Src = Op.getOperand(0);
3869 unsigned OpOpcode = Op.getOpcode();
3870 EVT SrcVT = Src.getValueType();
3871 EVT DstVT = Op.getValueType();
3872 SDValue SatVTOp = Op.getNode()->getOperand(1);
3873 EVT SatVT = cast<VTSDNode>(SatVTOp)->getVT();
3874 SDLoc DL(Op);
3875
3876 uint64_t DstWidth = DstVT.getScalarSizeInBits();
3877 uint64_t SatWidth = SatVT.getScalarSizeInBits();
3878 assert(SatWidth <= DstWidth && "Saturation width cannot exceed result width");
3879
3880 // Scalar cases will be selected natively to v_cvt_/s_cvt_ instructions.
3881 // v2f32 -> v2i16 will be selected natively to v_cvt_pk_[iu]16_f32.
3882 if (SatWidth == DstWidth) {
3883 if ((DstVT == MVT::i32 && (SrcVT == MVT::f32 || SrcVT == MVT::f64)) ||
3884 (DstVT == MVT::i16 && (SrcVT == MVT::f16 || SrcVT == MVT::f32)) ||
3885 (DstVT == MVT::v2i16 && SrcVT == MVT::v2f32))
3886 return Op;
3887 }
3888
3889 // Vectors can only be selected natively.
3890 if (DstVT.isVector())
3891 return SDValue();
3892
3893 // Perform all saturation at selected width (i16 or i32) and truncate
3894 if (SatWidth < DstWidth && SatWidth <= 32) {
3895 // For f16 conversion with sub-i16 saturation perform saturation
3896 // at i16, if available in the target. This removes the need for extra f16
3897 // to f32 conversion. For all the others use i32.
3898 MVT ResultVT =
3899 Subtarget->has16BitInsts() && SrcVT == MVT::f16 && SatWidth < 16
3900 ? MVT::i16
3901 : MVT::i32;
3902
3903 const SDValue ResultVTOp = DAG.getValueType(ResultVT);
3904 const uint64_t ResultWidth = ResultVT.getScalarSizeInBits();
3905
3906 // First, convert input float into selected integer (i16 or i32)
3907 SDValue FpToInt = DAG.getNode(OpOpcode, DL, ResultVT, Src, ResultVTOp);
3908 SDValue IntSatVal;
3909
3910 // Then, clamp at the saturation width using either i16 or i32 instructions
3911 if (OpOpcode == ISD::FP_TO_SINT_SAT) {
3912 SDValue MinConst = DAG.getConstant(
3913 APInt::getSignedMaxValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3914 SDValue MaxConst = DAG.getConstant(
3915 APInt::getSignedMinValue(SatWidth).sext(ResultWidth), DL, ResultVT);
3916 SDValue MinVal = DAG.getNode(ISD::SMIN, DL, ResultVT, FpToInt, MinConst);
3917 IntSatVal = DAG.getNode(ISD::SMAX, DL, ResultVT, MinVal, MaxConst);
3918 } else {
3919 SDValue MinConst = DAG.getConstant(
3920 APInt::getMaxValue(SatWidth).zext(ResultWidth), DL, ResultVT);
3921 IntSatVal = DAG.getNode(ISD::UMIN, DL, ResultVT, FpToInt, MinConst);
3922 }
3923
3924 // Finally, after saturating at i16 or i32 fit into the destination type
3925 return DAG.getExtOrTrunc(OpOpcode == ISD::FP_TO_SINT_SAT, IntSatVal, DL,
3926 DstVT);
3927 }
3928
3929 // SatWidth == DstWidth or SatWidth > 32
3930
3931 // Saturate at i32 for i64 dst and f16/bf16 src (will invoke f16 promotion
3932 // below)
3933 if (DstVT == MVT::i64 &&
3934 (SrcVT == MVT::f16 || SrcVT == MVT::bf16 ||
3935 (SrcVT == MVT::f32 && Src.getOpcode() == ISD::FP16_TO_FP))) {
3936 const SDValue Int32VTOp = DAG.getValueType(MVT::i32);
3937 return DAG.getNode(OpOpcode, DL, DstVT, Src, Int32VTOp);
3938 }
3939
3940 // Promote f16/bf16 src to f32 for i32 conversion
3941 if (DstVT == MVT::i32 && (SrcVT == MVT::f16 || SrcVT == MVT::bf16)) {
3942 SDValue PromotedSrc = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f32, Src);
3943 return DAG.getNode(Op.getOpcode(), DL, DstVT, PromotedSrc, SatVTOp);
3944 }
3945
3946 // For DstWidth < 16, promote i1 and i8 dst to i16 (if legal) with sub-i16
3947 // saturation. For DstWidth == 16, promote i16 dst to i32 with sub-i32
3948 // saturation; this covers i16.f32 and i16.f64
3949 if (DstWidth < 32) {
3950 // Note: this triggers SatWidth < DstWidth above to generate saturated
3951 // truncate by requesting MVT::i16/i32 destination with SatWidth < 16/32.
3952 MVT PromoteVT =
3953 (DstWidth < 16 && Subtarget->has16BitInsts()) ? MVT::i16 : MVT::i32;
3954 SDValue FpToInt = DAG.getNode(OpOpcode, DL, PromoteVT, Src, SatVTOp);
3955 return DAG.getNode(ISD::TRUNCATE, DL, DstVT, FpToInt);
3956 }
3957
3958 // TODO: can we implement i64 dst for f32/f64?
3959
3960 return SDValue();
3961}
3962
3964 SelectionDAG &DAG) const {
3965 EVT ExtraVT = cast<VTSDNode>(Op.getOperand(1))->getVT();
3966 MVT VT = Op.getSimpleValueType();
3967 MVT ScalarVT = VT.getScalarType();
3968
3969 assert(VT.isVector());
3970
3971 SDValue Src = Op.getOperand(0);
3972 SDLoc DL(Op);
3973
3974 // TODO: Don't scalarize on Evergreen?
3975 unsigned NElts = VT.getVectorNumElements();
3977 DAG.ExtractVectorElements(Src, Args, 0, NElts);
3978
3979 SDValue VTOp = DAG.getValueType(ExtraVT.getScalarType());
3980 for (unsigned I = 0; I < NElts; ++I)
3981 Args[I] = DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, ScalarVT, Args[I], VTOp);
3982
3983 return DAG.getBuildVector(VT, DL, Args);
3984}
3985
3986//===----------------------------------------------------------------------===//
3987// Custom DAG optimizations
3988//===----------------------------------------------------------------------===//
3989
3990static bool isU24(SDValue Op, SelectionDAG &DAG) {
3991 return AMDGPUTargetLowering::numBitsUnsigned(Op, DAG) <= 24;
3992}
3993
3994static bool isI24(SDValue Op, SelectionDAG &DAG) {
3995 EVT VT = Op.getValueType();
3996 return VT.getSizeInBits() >= 24 && // Types less than 24-bit should be treated
3997 // as unsigned 24-bit values.
3999}
4000
4003 SelectionDAG &DAG = DCI.DAG;
4004 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
4005 bool IsIntrin = Node24->getOpcode() == ISD::INTRINSIC_WO_CHAIN;
4006
4007 SDValue LHS = IsIntrin ? Node24->getOperand(1) : Node24->getOperand(0);
4008 SDValue RHS = IsIntrin ? Node24->getOperand(2) : Node24->getOperand(1);
4009 unsigned NewOpcode = Node24->getOpcode();
4010 if (IsIntrin) {
4011 unsigned IID = Node24->getConstantOperandVal(0);
4012 switch (IID) {
4013 case Intrinsic::amdgcn_mul_i24:
4014 NewOpcode = AMDGPUISD::MUL_I24;
4015 break;
4016 case Intrinsic::amdgcn_mul_u24:
4017 NewOpcode = AMDGPUISD::MUL_U24;
4018 break;
4019 case Intrinsic::amdgcn_mulhi_i24:
4020 NewOpcode = AMDGPUISD::MULHI_I24;
4021 break;
4022 case Intrinsic::amdgcn_mulhi_u24:
4023 NewOpcode = AMDGPUISD::MULHI_U24;
4024 break;
4025 default:
4026 llvm_unreachable("Expected 24-bit mul intrinsic");
4027 }
4028 }
4029
4030 APInt Demanded = APInt::getLowBitsSet(LHS.getValueSizeInBits(), 24);
4031
4032 // First try to simplify using SimplifyMultipleUseDemandedBits which allows
4033 // the operands to have other uses, but will only perform simplifications that
4034 // involve bypassing some nodes for this user.
4035 SDValue DemandedLHS = TLI.SimplifyMultipleUseDemandedBits(LHS, Demanded, DAG);
4036 SDValue DemandedRHS = TLI.SimplifyMultipleUseDemandedBits(RHS, Demanded, DAG);
4037 if (DemandedLHS || DemandedRHS)
4038 return DAG.getNode(NewOpcode, SDLoc(Node24), Node24->getVTList(),
4039 DemandedLHS ? DemandedLHS : LHS,
4040 DemandedRHS ? DemandedRHS : RHS);
4041
4042 // Now try SimplifyDemandedBits which can simplify the nodes used by our
4043 // operands if this node is the only user.
4044 if (TLI.SimplifyDemandedBits(LHS, Demanded, DCI))
4045 return SDValue(Node24, 0);
4046 if (TLI.SimplifyDemandedBits(RHS, Demanded, DCI))
4047 return SDValue(Node24, 0);
4048
4049 return SDValue();
4050}
4051
4052template <typename IntTy>
4054 uint32_t Width, const SDLoc &DL) {
4055 if (Width + Offset < 32) {
4056 uint32_t Shl = static_cast<uint32_t>(Src0) << (32 - Offset - Width);
4057 IntTy Result = static_cast<IntTy>(Shl) >> (32 - Width);
4058 if constexpr (std::is_signed_v<IntTy>) {
4059 return DAG.getSignedConstant(Result, DL, MVT::i32);
4060 } else {
4061 return DAG.getConstant(Result, DL, MVT::i32);
4062 }
4063 }
4064
4065 return DAG.getConstant(Src0 >> Offset, DL, MVT::i32);
4066}
4067
4068static bool hasVolatileUser(SDNode *Val) {
4069 for (SDNode *U : Val->users()) {
4070 if (MemSDNode *M = dyn_cast<MemSDNode>(U)) {
4071 if (M->isVolatile())
4072 return true;
4073 }
4074 }
4075
4076 return false;
4077}
4078
4080 // i32 vectors are the canonical memory type.
4081 if (VT.getScalarType() == MVT::i32 || isTypeLegal(VT))
4082 return false;
4083
4084 if (!VT.isByteSized())
4085 return false;
4086
4087 unsigned Size = VT.getStoreSize();
4088
4089 if ((Size == 1 || Size == 2 || Size == 4) && !VT.isVector())
4090 return false;
4091
4092 if (Size == 3 || (Size > 4 && (Size % 4 != 0)))
4093 return false;
4094
4095 return true;
4096}
4097
4098// Replace load of an illegal type with a bitcast from a load of a friendlier
4099// type.
4101 DAGCombinerInfo &DCI) const {
4102 if (!DCI.isBeforeLegalize())
4103 return SDValue();
4104
4106 if (!LN->isSimple() || !ISD::isNormalLoad(LN) || hasVolatileUser(LN))
4107 return SDValue();
4108
4109 SDLoc SL(N);
4110 SelectionDAG &DAG = DCI.DAG;
4111 EVT VT = LN->getMemoryVT();
4112
4113 unsigned Size = VT.getStoreSize();
4114 Align Alignment = LN->getAlign();
4115 if (Alignment < Size && isTypeLegal(VT)) {
4116 unsigned IsFast;
4117 unsigned AS = LN->getAddressSpace();
4118
4119 // Expand unaligned loads earlier than legalization. Due to visitation order
4120 // problems during legalization, the emitted instructions to pack and unpack
4121 // the bytes again are not eliminated in the case of an unaligned copy.
4123 VT, AS, Alignment, LN->getMemOperand()->getFlags(), &IsFast)) {
4124 if (VT.isVector())
4125 return SplitVectorLoad(SDValue(LN, 0), DAG);
4126
4127 SDValue Ops[2];
4128 std::tie(Ops[0], Ops[1]) = expandUnalignedLoad(LN, DAG);
4129
4130 return DAG.getMergeValues(Ops, SDLoc(N));
4131 }
4132
4133 if (!IsFast)
4134 return SDValue();
4135 }
4136
4137 if (!shouldCombineMemoryType(VT))
4138 return SDValue();
4139
4140 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4141
4142 SDValue NewLoad
4143 = DAG.getLoad(NewVT, SL, LN->getChain(),
4144 LN->getBasePtr(), LN->getMemOperand());
4145
4146 SDValue BC = DAG.getNode(ISD::BITCAST, SL, VT, NewLoad);
4147 DCI.CombineTo(N, BC, NewLoad.getValue(1));
4148 return SDValue(N, 0);
4149}
4150
4151// Replace store of an illegal type with a store of a bitcast to a friendlier
4152// type.
4154 DAGCombinerInfo &DCI) const {
4155 if (!DCI.isBeforeLegalize())
4156 return SDValue();
4157
4159 if (!SN->isSimple() || !ISD::isNormalStore(SN))
4160 return SDValue();
4161
4162 EVT VT = SN->getMemoryVT();
4163 unsigned Size = VT.getStoreSize();
4164
4165 SDLoc SL(N);
4166 SelectionDAG &DAG = DCI.DAG;
4167 Align Alignment = SN->getAlign();
4168 if (Alignment < Size && isTypeLegal(VT)) {
4169 unsigned IsFast;
4170 unsigned AS = SN->getAddressSpace();
4171
4172 // Expand unaligned stores earlier than legalization. Due to visitation
4173 // order problems during legalization, the emitted instructions to pack and
4174 // unpack the bytes again are not eliminated in the case of an unaligned
4175 // copy.
4177 VT, AS, Alignment, SN->getMemOperand()->getFlags(), &IsFast)) {
4178 if (VT.isVector())
4179 return SplitVectorStore(SDValue(SN, 0), DAG);
4180
4181 return expandUnalignedStore(SN, DAG);
4182 }
4183
4184 if (!IsFast)
4185 return SDValue();
4186 }
4187
4188 if (!shouldCombineMemoryType(VT))
4189 return SDValue();
4190
4191 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
4192 SDValue Val = SN->getValue();
4193
4194 // DCI.AddToWorklist(Val.getNode());
4195
4196 bool OtherUses = !Val.hasOneUse();
4197 SDValue CastVal = DAG.getBitcast(NewVT, Val);
4198 if (OtherUses) {
4199 SDValue CastBack = DAG.getBitcast(VT, CastVal);
4200 DAG.ReplaceAllUsesOfValueWith(Val, CastBack);
4201 }
4202
4203 return DAG.getStore(SN->getChain(), SL, CastVal,
4204 SN->getBasePtr(), SN->getMemOperand());
4205}
4206
4207// FIXME: This should go in generic DAG combiner with an isTruncateFree check,
4208// but isTruncateFree is inaccurate for i16 now because of SALU vs. VALU
4209// issues.
4211 DAGCombinerInfo &DCI) const {
4212 SelectionDAG &DAG = DCI.DAG;
4213 SDValue N0 = N->getOperand(0);
4214
4215 // (vt2 (assertzext (truncate vt0:x), vt1)) ->
4216 // (vt2 (truncate (assertzext vt0:x, vt1)))
4217 if (N0.getOpcode() == ISD::TRUNCATE) {
4218 SDValue N1 = N->getOperand(1);
4219 EVT ExtVT = cast<VTSDNode>(N1)->getVT();
4220 SDLoc SL(N);
4221
4222 SDValue Src = N0.getOperand(0);
4223 EVT SrcVT = Src.getValueType();
4224 if (SrcVT.bitsGE(ExtVT)) {
4225 SDValue NewInReg = DAG.getNode(N->getOpcode(), SL, SrcVT, Src, N1);
4226 return DAG.getNode(ISD::TRUNCATE, SL, N->getValueType(0), NewInReg);
4227 }
4228 }
4229
4230 return SDValue();
4231}
4232
4234 SDNode *N, DAGCombinerInfo &DCI) const {
4235 unsigned IID = N->getConstantOperandVal(0);
4236 switch (IID) {
4237 case Intrinsic::amdgcn_mul_i24:
4238 case Intrinsic::amdgcn_mul_u24:
4239 case Intrinsic::amdgcn_mulhi_i24:
4240 case Intrinsic::amdgcn_mulhi_u24:
4241 return simplifyMul24(N, DCI);
4242 case Intrinsic::amdgcn_fract:
4243 case Intrinsic::amdgcn_rsq:
4244 case Intrinsic::amdgcn_rcp_legacy:
4245 case Intrinsic::amdgcn_rsq_legacy:
4246 case Intrinsic::amdgcn_rsq_clamp:
4247 case Intrinsic::amdgcn_tanh:
4248 case Intrinsic::amdgcn_prng_b32: {
4249 // FIXME: This is probably wrong. If src is an sNaN, it won't be quieted
4250 SDValue Src = N->getOperand(1);
4251 return Src.isUndef() ? Src : SDValue();
4252 }
4253 case Intrinsic::amdgcn_frexp_exp: {
4254 // frexp_exp (fneg x) -> frexp_exp x
4255 // frexp_exp (fabs x) -> frexp_exp x
4256 // frexp_exp (fneg (fabs x)) -> frexp_exp x
4257 SDValue Src = N->getOperand(1);
4258 SDValue PeekSign = peekFPSignOps(Src);
4259 if (PeekSign == Src)
4260 return SDValue();
4261 return SDValue(DCI.DAG.UpdateNodeOperands(N, N->getOperand(0), PeekSign),
4262 0);
4263 }
4264 default:
4265 return SDValue();
4266 }
4267}
4268
4269/// Split the 64-bit value \p LHS into two 32-bit components, and perform the
4270/// binary operation \p Opc to it with the corresponding constant operands.
4272 DAGCombinerInfo &DCI, const SDLoc &SL,
4273 unsigned Opc, SDValue LHS,
4274 uint32_t ValLo, uint32_t ValHi) const {
4275 SelectionDAG &DAG = DCI.DAG;
4276 SDValue Lo, Hi;
4277 std::tie(Lo, Hi) = split64BitValue(LHS, DAG);
4278
4279 SDValue LoRHS = DAG.getConstant(ValLo, SL, MVT::i32);
4280 SDValue HiRHS = DAG.getConstant(ValHi, SL, MVT::i32);
4281
4282 SDValue LoAnd = DAG.getNode(Opc, SL, MVT::i32, Lo, LoRHS);
4283 SDValue HiAnd = DAG.getNode(Opc, SL, MVT::i32, Hi, HiRHS);
4284
4285 // Re-visit the ands. It's possible we eliminated one of them and it could
4286 // simplify the vector.
4287 DCI.AddToWorklist(Lo.getNode());
4288 DCI.AddToWorklist(Hi.getNode());
4289
4290 SDValue Vec = DAG.getBuildVector(MVT::v2i32, SL, {LoAnd, HiAnd});
4291 return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
4292}
4293
4295 DAGCombinerInfo &DCI) const {
4296 EVT VT = N->getValueType(0);
4297 SDValue LHS = N->getOperand(0);
4298 SDValue RHS = N->getOperand(1);
4300 SDLoc SL(N);
4301 SelectionDAG &DAG = DCI.DAG;
4302
4303 unsigned RHSVal;
4304 if (CRHS) {
4305 RHSVal = CRHS->getZExtValue();
4306 if (!RHSVal)
4307 return LHS;
4308
4309 switch (LHS->getOpcode()) {
4310 default:
4311 break;
4312 case ISD::ZERO_EXTEND:
4313 case ISD::SIGN_EXTEND:
4314 case ISD::ANY_EXTEND: {
4315 SDValue X = LHS->getOperand(0);
4316
4317 if (VT == MVT::i32 && RHSVal == 16 && X.getValueType() == MVT::i16 &&
4318 isOperationLegal(ISD::BUILD_VECTOR, MVT::v2i16)) {
4319 // Prefer build_vector as the canonical form if packed types are legal.
4320 // (shl ([asz]ext i16:x), 16 -> build_vector 0, x
4321 SDValue Vec = DAG.getBuildVector(
4322 MVT::v2i16, SL,
4323 {DAG.getConstant(0, SL, MVT::i16), LHS->getOperand(0)});
4324 return DAG.getNode(ISD::BITCAST, SL, MVT::i32, Vec);
4325 }
4326
4327 // shl (ext x) => zext (shl x), if shift does not overflow int
4328 if (VT != MVT::i64)
4329 break;
4331 unsigned LZ = Known.countMinLeadingZeros();
4332 if (LZ < RHSVal)
4333 break;
4334 EVT XVT = X.getValueType();
4335 SDValue Shl = DAG.getNode(ISD::SHL, SL, XVT, X, SDValue(CRHS, 0));
4336 return DAG.getZExtOrTrunc(Shl, SL, VT);
4337 }
4338 }
4339 }
4340
4341 if (VT.getScalarType() != MVT::i64)
4342 return SDValue();
4343
4344 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4345 // common case, splitting this into a move and a 32-bit shift is faster and
4346 // the same code size.
4347 KnownBits Known = DAG.computeKnownBits(RHS);
4348
4349 EVT ElementType = VT.getScalarType();
4350 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4351 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4352
4353 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4354 return SDValue();
4355 SDValue ShiftAmt;
4356
4357 if (CRHS) {
4358 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4359 TargetType);
4360 } else {
4361 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4362 const SDValue ShiftMask =
4363 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4364 // This AND instruction will clamp out of bounds shift values.
4365 // It will also be removed during later instruction selection.
4366 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4367 }
4368
4369 SDValue Lo = DAG.getNode(ISD::TRUNCATE, SL, TargetType, LHS);
4370 SDValue NewShift =
4371 DAG.getNode(ISD::SHL, SL, TargetType, Lo, ShiftAmt, N->getFlags());
4372
4373 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4374 SDValue Vec;
4375
4376 if (VT.isVector()) {
4377 EVT ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4378 unsigned NElts = TargetType.getVectorNumElements();
4380 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4381
4382 DAG.ExtractVectorElements(NewShift, HiOps, 0, NElts);
4383 for (unsigned I = 0; I != NElts; ++I)
4384 HiAndLoOps[2 * I + 1] = HiOps[I];
4385 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4386 } else {
4387 EVT ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4388 Vec = DAG.getBuildVector(ConcatType, SL, {Zero, NewShift});
4389 }
4390 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4391}
4392
4394 DAGCombinerInfo &DCI) const {
4395 SDValue RHS = N->getOperand(1);
4397 EVT VT = N->getValueType(0);
4398 SDValue LHS = N->getOperand(0);
4399 SelectionDAG &DAG = DCI.DAG;
4400 SDLoc SL(N);
4401
4402 if (VT.getScalarType() != MVT::i64)
4403 return SDValue();
4404
4405 // For C >= 32
4406 // i64 (sra x, C) -> (build_pair (sra hi_32(x), C - 32), sra hi_32(x), 31))
4407
4408 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4409 // common case, splitting this into a move and a 32-bit shift is faster and
4410 // the same code size.
4411 KnownBits Known = DAG.computeKnownBits(RHS);
4412
4413 EVT ElementType = VT.getScalarType();
4414 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4415 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4416
4417 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4418 return SDValue();
4419
4420 SDValue ShiftFullAmt =
4421 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4422 SDValue ShiftAmt;
4423 if (CRHS) {
4424 unsigned RHSVal = CRHS->getZExtValue();
4425 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4426 TargetType);
4427 } else if (Known.getMinValue().getZExtValue() ==
4428 (ElementType.getSizeInBits() - 1)) {
4429 ShiftAmt = ShiftFullAmt;
4430 } else {
4431 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4432 const SDValue ShiftMask =
4433 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4434 // This AND instruction will clamp out of bounds shift values.
4435 // It will also be removed during later instruction selection.
4436 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4437 }
4438
4439 EVT ConcatType;
4440 SDValue Hi;
4441 SDLoc LHSSL(LHS);
4442 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4443 if (VT.isVector()) {
4444 unsigned NElts = TargetType.getVectorNumElements();
4445 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4446 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4447 SmallVector<SDValue, 8> HiOps(NElts);
4448 SmallVector<SDValue, 16> HiAndLoOps;
4449
4450 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, 0, NElts * 2);
4451 for (unsigned I = 0; I != NElts; ++I) {
4452 HiOps[I] = HiAndLoOps[2 * I + 1];
4453 }
4454 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4455 } else {
4456 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4457 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4458 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4459 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4460 }
4461
4462 KnownBits KnownLHS = DAG.computeKnownBits(LHS);
4463 SDValue NewShift, HiShift;
4464 if (KnownLHS.isNegative()) {
4465 HiShift = DAG.getAllOnesConstant(SL, TargetType);
4466 NewShift =
4467 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4468 } else if (CRHS &&
4469 CRHS->getZExtValue() == (ElementType.getSizeInBits() - 1)) {
4470 NewShift = HiShift =
4471 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4472 } else {
4473 Hi = DAG.getFreeze(Hi);
4474 HiShift = DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftFullAmt);
4475 NewShift =
4476 DAG.getNode(ISD::SRA, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4477 }
4478
4479 SDValue Vec;
4480 if (VT.isVector()) {
4481 unsigned NElts = TargetType.getVectorNumElements();
4484 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2);
4485
4486 DAG.ExtractVectorElements(HiShift, HiOps, 0, NElts);
4487 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4488 for (unsigned I = 0; I != NElts; ++I) {
4489 HiAndLoOps[2 * I + 1] = HiOps[I];
4490 HiAndLoOps[2 * I] = LoOps[I];
4491 }
4492 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4493 } else {
4494 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, HiShift});
4495 }
4496 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4497}
4498
4500 DAGCombinerInfo &DCI) const {
4501 SDValue RHS = N->getOperand(1);
4503 EVT VT = N->getValueType(0);
4504 SDValue LHS = N->getOperand(0);
4505 SelectionDAG &DAG = DCI.DAG;
4506 SDLoc SL(N);
4507 unsigned RHSVal;
4508
4509 if (CRHS) {
4510 RHSVal = CRHS->getZExtValue();
4511
4512 // fold (srl (and x, c1 << c2), c2) -> (and (srl(x, c2), c1)
4513 // this improves the ability to match BFE patterns in isel.
4514 if (LHS.getOpcode() == ISD::AND) {
4515 if (auto *Mask = dyn_cast<ConstantSDNode>(LHS.getOperand(1))) {
4516 unsigned MaskIdx, MaskLen;
4517 if (Mask->getAPIntValue().isShiftedMask(MaskIdx, MaskLen) &&
4518 MaskIdx == RHSVal) {
4519 return DAG.getNode(ISD::AND, SL, VT,
4520 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(0),
4521 N->getOperand(1)),
4522 DAG.getNode(ISD::SRL, SL, VT, LHS.getOperand(1),
4523 N->getOperand(1)));
4524 }
4525 }
4526 }
4527 }
4528
4529 if (VT.getScalarType() != MVT::i64)
4530 return SDValue();
4531
4532 // for C >= 32
4533 // i64 (srl x, C) -> (build_pair (srl hi_32(x), C - 32), 0)
4534
4535 // On some subtargets, 64-bit shift is a quarter rate instruction. In the
4536 // common case, splitting this into a move and a 32-bit shift is faster and
4537 // the same code size.
4538 KnownBits Known = DAG.computeKnownBits(RHS);
4539
4540 EVT ElementType = VT.getScalarType();
4541 EVT TargetScalarType = ElementType.getHalfSizedIntegerVT(*DAG.getContext());
4542 EVT TargetType = VT.changeElementType(*DAG.getContext(), TargetScalarType);
4543
4544 if (Known.getMinValue().getZExtValue() < TargetScalarType.getSizeInBits())
4545 return SDValue();
4546
4547 SDValue ShiftAmt;
4548 if (CRHS) {
4549 ShiftAmt = DAG.getConstant(RHSVal - TargetScalarType.getSizeInBits(), SL,
4550 TargetType);
4551 } else {
4552 SDValue TruncShiftAmt = DAG.getNode(ISD::TRUNCATE, SL, TargetType, RHS);
4553 const SDValue ShiftMask =
4554 DAG.getConstant(TargetScalarType.getSizeInBits() - 1, SL, TargetType);
4555 // This AND instruction will clamp out of bounds shift values.
4556 // It will also be removed during later instruction selection.
4557 ShiftAmt = DAG.getNode(ISD::AND, SL, TargetType, TruncShiftAmt, ShiftMask);
4558 }
4559
4560 const SDValue Zero = DAG.getConstant(0, SL, TargetScalarType);
4561 EVT ConcatType;
4562 SDValue Hi;
4563 SDLoc LHSSL(LHS);
4564 // Bitcast LHS into ConcatType so hi-half of source can be extracted into Hi
4565 if (VT.isVector()) {
4566 unsigned NElts = TargetType.getVectorNumElements();
4567 ConcatType = TargetType.getDoubleNumVectorElementsVT(*DAG.getContext());
4568 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4569 SmallVector<SDValue, 8> HiOps(NElts);
4570 SmallVector<SDValue, 16> HiAndLoOps;
4571
4572 DAG.ExtractVectorElements(SplitLHS, HiAndLoOps, /*Start=*/0, NElts * 2);
4573 for (unsigned I = 0; I != NElts; ++I)
4574 HiOps[I] = HiAndLoOps[2 * I + 1];
4575 Hi = DAG.getNode(ISD::BUILD_VECTOR, LHSSL, TargetType, HiOps);
4576 } else {
4577 const SDValue One = DAG.getConstant(1, LHSSL, TargetScalarType);
4578 ConcatType = EVT::getVectorVT(*DAG.getContext(), TargetType, 2);
4579 SDValue SplitLHS = DAG.getNode(ISD::BITCAST, LHSSL, ConcatType, LHS);
4580 Hi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, LHSSL, TargetType, SplitLHS, One);
4581 }
4582
4583 SDValue NewShift =
4584 DAG.getNode(ISD::SRL, SL, TargetType, Hi, ShiftAmt, N->getFlags());
4585
4586 SDValue Vec;
4587 if (VT.isVector()) {
4588 unsigned NElts = TargetType.getVectorNumElements();
4590 SmallVector<SDValue, 16> HiAndLoOps(NElts * 2, Zero);
4591
4592 DAG.ExtractVectorElements(NewShift, LoOps, 0, NElts);
4593 for (unsigned I = 0; I != NElts; ++I)
4594 HiAndLoOps[2 * I] = LoOps[I];
4595 Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, ConcatType, HiAndLoOps);
4596 } else {
4597 Vec = DAG.getBuildVector(ConcatType, SL, {NewShift, Zero});
4598 }
4599 return DAG.getNode(ISD::BITCAST, SL, VT, Vec);
4600}
4601
4603 SDNode *N, DAGCombinerInfo &DCI) const {
4604 SDLoc SL(N);
4605 SelectionDAG &DAG = DCI.DAG;
4606 EVT VT = N->getValueType(0);
4607 SDValue Src = N->getOperand(0);
4608
4609 // vt1 (truncate (bitcast (build_vector vt0:x, ...))) -> vt1 (bitcast vt0:x)
4610 if (Src.getOpcode() == ISD::BITCAST && !VT.isVector()) {
4611 SDValue Vec = Src.getOperand(0);
4612 if (Vec.getOpcode() == ISD::BUILD_VECTOR) {
4613 SDValue Elt0 = Vec.getOperand(0);
4614 EVT EltVT = Elt0.getValueType();
4615 if (VT.getFixedSizeInBits() <= EltVT.getFixedSizeInBits()) {
4616 if (EltVT.isFloatingPoint()) {
4617 Elt0 = DAG.getNode(ISD::BITCAST, SL,
4618 EltVT.changeTypeToInteger(), Elt0);
4619 }
4620
4621 return DAG.getNode(ISD::TRUNCATE, SL, VT, Elt0);
4622 }
4623 }
4624 }
4625
4626 // Equivalent of above for accessing the high element of a vector as an
4627 // integer operation.
4628 // trunc (srl (bitcast (build_vector x, y))), 16 -> trunc (bitcast y)
4629 if (Src.getOpcode() == ISD::SRL && !VT.isVector()) {
4630 if (auto *K = isConstOrConstSplat(Src.getOperand(1))) {
4631 SDValue BV = stripBitcast(Src.getOperand(0));
4632 if (BV.getOpcode() == ISD::BUILD_VECTOR) {
4633 EVT SrcEltVT = BV.getOperand(0).getValueType();
4634 unsigned SrcEltSize = SrcEltVT.getSizeInBits();
4635 unsigned BitIndex = K->getZExtValue();
4636 unsigned PartIndex = BitIndex / SrcEltSize;
4637
4638 if (PartIndex * SrcEltSize == BitIndex &&
4639 PartIndex < BV.getNumOperands()) {
4640 if (SrcEltVT.getSizeInBits() == VT.getSizeInBits()) {
4641 SDValue SrcElt =
4642 DAG.getNode(ISD::BITCAST, SL, SrcEltVT.changeTypeToInteger(),
4643 BV.getOperand(PartIndex));
4644 return DAG.getNode(ISD::TRUNCATE, SL, VT, SrcElt);
4645 }
4646 }
4647 }
4648 }
4649 }
4650
4651 // Partially shrink 64-bit shifts to 32-bit if reduced to 16-bit.
4652 //
4653 // i16 (trunc (srl i64:x, K)), K <= 16 ->
4654 // i16 (trunc (srl (i32 (trunc x), K)))
4655 if (VT.getScalarSizeInBits() < 32) {
4656 EVT SrcVT = Src.getValueType();
4657 if (SrcVT.getScalarSizeInBits() > 32 &&
4658 (Src.getOpcode() == ISD::SRL ||
4659 Src.getOpcode() == ISD::SRA ||
4660 Src.getOpcode() == ISD::SHL)) {
4661 SDValue Amt = Src.getOperand(1);
4662 KnownBits Known = DAG.computeKnownBits(Amt);
4663
4664 // - For left shifts, do the transform as long as the shift
4665 // amount is still legal for i32, so when ShiftAmt < 32 (<= 31)
4666 // - For right shift, do it if ShiftAmt <= (32 - Size) to avoid
4667 // losing information stored in the high bits when truncating.
4668 const unsigned MaxCstSize =
4669 (Src.getOpcode() == ISD::SHL) ? 31 : (32 - VT.getScalarSizeInBits());
4670 if (Known.getMaxValue().ule(MaxCstSize)) {
4671 EVT MidVT = VT.isVector() ?
4672 EVT::getVectorVT(*DAG.getContext(), MVT::i32,
4673 VT.getVectorNumElements()) : MVT::i32;
4674
4675 EVT NewShiftVT = getShiftAmountTy(MidVT, DAG.getDataLayout());
4676 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, MidVT,
4677 Src.getOperand(0));
4678 DCI.AddToWorklist(Trunc.getNode());
4679
4680 if (Amt.getValueType() != NewShiftVT) {
4681 Amt = DAG.getZExtOrTrunc(Amt, SL, NewShiftVT);
4682 DCI.AddToWorklist(Amt.getNode());
4683 }
4684
4685 SDValue ShrunkShift = DAG.getNode(Src.getOpcode(), SL, MidVT,
4686 Trunc, Amt);
4687 return DAG.getNode(ISD::TRUNCATE, SL, VT, ShrunkShift);
4688 }
4689 }
4690 }
4691
4692 return SDValue();
4693}
4694
4695// We need to specifically handle i64 mul here to avoid unnecessary conversion
4696// instructions. If we only match on the legalized i64 mul expansion,
4697// SimplifyDemandedBits will be unable to remove them because there will be
4698// multiple uses due to the separate mul + mulh[su].
4699static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL,
4700 SDValue N0, SDValue N1, unsigned Size, bool Signed) {
4701 if (Size <= 32) {
4702 unsigned MulOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4703 return DAG.getNode(MulOpc, SL, MVT::i32, N0, N1);
4704 }
4705
4706 unsigned MulLoOpc = Signed ? AMDGPUISD::MUL_I24 : AMDGPUISD::MUL_U24;
4707 unsigned MulHiOpc = Signed ? AMDGPUISD::MULHI_I24 : AMDGPUISD::MULHI_U24;
4708
4709 SDValue MulLo = DAG.getNode(MulLoOpc, SL, MVT::i32, N0, N1);
4710 SDValue MulHi = DAG.getNode(MulHiOpc, SL, MVT::i32, N0, N1);
4711
4712 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i64, MulLo, MulHi);
4713}
4714
4715/// If \p V is an add of a constant 1, returns the other operand. Otherwise
4716/// return SDValue().
4717static SDValue getAddOneOp(const SDNode *V) {
4718 if (V->getOpcode() != ISD::ADD)
4719 return SDValue();
4720
4721 return isOneConstant(V->getOperand(1)) ? V->getOperand(0) : SDValue();
4722}
4723
4725 DAGCombinerInfo &DCI) const {
4726 assert(N->getOpcode() == ISD::MUL);
4727 EVT VT = N->getValueType(0);
4728
4729 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4730 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4731 // unnecessarily). isDivergent() is used as an approximation of whether the
4732 // value is in an SGPR.
4733 if (!N->isDivergent())
4734 return SDValue();
4735
4736 unsigned Size = VT.getSizeInBits();
4737 if (VT.isVector() || Size > 64)
4738 return SDValue();
4739
4740 SelectionDAG &DAG = DCI.DAG;
4741 SDLoc DL(N);
4742
4743 SDValue N0 = N->getOperand(0);
4744 SDValue N1 = N->getOperand(1);
4745
4746 // Undo InstCombine canonicalize X * (Y + 1) -> X * Y + X to enable mad
4747 // matching.
4748
4749 // mul x, (add y, 1) -> add (mul x, y), x
4750 auto IsFoldableAdd = [](SDValue V) -> SDValue {
4751 SDValue AddOp = getAddOneOp(V.getNode());
4752 if (!AddOp)
4753 return SDValue();
4754
4755 if (V.hasOneUse() || all_of(V->users(), [](const SDNode *U) -> bool {
4756 return U->getOpcode() == ISD::MUL;
4757 }))
4758 return AddOp;
4759
4760 return SDValue();
4761 };
4762
4763 // FIXME: The selection pattern is not properly checking for commuted
4764 // operands, so we have to place the mul in the LHS
4765 if (SDValue MulOper = IsFoldableAdd(N0)) {
4766 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N1, MulOper);
4767 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N1);
4768 }
4769
4770 if (SDValue MulOper = IsFoldableAdd(N1)) {
4771 SDValue MulVal = DAG.getNode(N->getOpcode(), DL, VT, N0, MulOper);
4772 return DAG.getNode(ISD::ADD, DL, VT, MulVal, N0);
4773 }
4774
4775 // There are i16 integer mul/mad.
4776 if (isTypeLegal(MVT::i16) && VT.getScalarType().bitsLE(MVT::i16))
4777 return SDValue();
4778
4779 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4780 // in the source into any_extends if the result of the mul is truncated. Since
4781 // we can assume the high bits are whatever we want, use the underlying value
4782 // to avoid the unknown high bits from interfering.
4783 if (N0.getOpcode() == ISD::ANY_EXTEND)
4784 N0 = N0.getOperand(0);
4785
4786 if (N1.getOpcode() == ISD::ANY_EXTEND)
4787 N1 = N1.getOperand(0);
4788
4789 SDValue Mul;
4790
4791 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4792 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4793 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4794 Mul = getMul24(DAG, DL, N0, N1, Size, false);
4795 } else if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4796 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4797 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4798 Mul = getMul24(DAG, DL, N0, N1, Size, true);
4799 } else {
4800 return SDValue();
4801 }
4802
4803 // We need to use sext even for MUL_U24, because MUL_U24 is used
4804 // for signed multiply of 8 and 16-bit types.
4805 return DAG.getSExtOrTrunc(Mul, DL, VT);
4806}
4807
4808SDValue
4810 DAGCombinerInfo &DCI) const {
4811 if (N->getValueType(0) != MVT::i32)
4812 return SDValue();
4813
4814 SelectionDAG &DAG = DCI.DAG;
4815 SDLoc DL(N);
4816
4817 bool Signed = N->getOpcode() == ISD::SMUL_LOHI;
4818 SDValue N0 = N->getOperand(0);
4819 SDValue N1 = N->getOperand(1);
4820
4821 // SimplifyDemandedBits has the annoying habit of turning useful zero_extends
4822 // in the source into any_extends if the result of the mul is truncated. Since
4823 // we can assume the high bits are whatever we want, use the underlying value
4824 // to avoid the unknown high bits from interfering.
4825 if (N0.getOpcode() == ISD::ANY_EXTEND)
4826 N0 = N0.getOperand(0);
4827 if (N1.getOpcode() == ISD::ANY_EXTEND)
4828 N1 = N1.getOperand(0);
4829
4830 // Try to use two fast 24-bit multiplies (one for each half of the result)
4831 // instead of one slow extending multiply.
4832 unsigned LoOpcode = 0;
4833 unsigned HiOpcode = 0;
4834 if (Signed) {
4835 if (Subtarget->hasMulI24() && isI24(N0, DAG) && isI24(N1, DAG)) {
4836 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4837 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4838 LoOpcode = AMDGPUISD::MUL_I24;
4839 HiOpcode = AMDGPUISD::MULHI_I24;
4840 }
4841 } else {
4842 if (Subtarget->hasMulU24() && isU24(N0, DAG) && isU24(N1, DAG)) {
4843 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4844 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4845 LoOpcode = AMDGPUISD::MUL_U24;
4846 HiOpcode = AMDGPUISD::MULHI_U24;
4847 }
4848 }
4849 if (!LoOpcode)
4850 return SDValue();
4851
4852 SDValue Lo = DAG.getNode(LoOpcode, DL, MVT::i32, N0, N1);
4853 SDValue Hi = DAG.getNode(HiOpcode, DL, MVT::i32, N0, N1);
4854 DCI.CombineTo(N, Lo, Hi);
4855 return SDValue(N, 0);
4856}
4857
4859 DAGCombinerInfo &DCI) const {
4860 EVT VT = N->getValueType(0);
4861
4862 if (!Subtarget->hasMulI24() || VT.isVector())
4863 return SDValue();
4864
4865 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4866 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4867 // unnecessarily). isDivergent() is used as an approximation of whether the
4868 // value is in an SGPR.
4869 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4870 // valu op anyway)
4871 if (Subtarget->hasSMulHi() && !N->isDivergent())
4872 return SDValue();
4873
4874 SelectionDAG &DAG = DCI.DAG;
4875 SDLoc DL(N);
4876
4877 SDValue N0 = N->getOperand(0);
4878 SDValue N1 = N->getOperand(1);
4879
4880 if (!isI24(N0, DAG) || !isI24(N1, DAG))
4881 return SDValue();
4882
4883 N0 = DAG.getSExtOrTrunc(N0, DL, MVT::i32);
4884 N1 = DAG.getSExtOrTrunc(N1, DL, MVT::i32);
4885
4886 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_I24, DL, MVT::i32, N0, N1);
4887 DCI.AddToWorklist(Mulhi.getNode());
4888 return DAG.getSExtOrTrunc(Mulhi, DL, VT);
4889}
4890
4892 DAGCombinerInfo &DCI) const {
4893 EVT VT = N->getValueType(0);
4894
4895 if (VT.isVector() || VT.getSizeInBits() > 32 || !Subtarget->hasMulU24())
4896 return SDValue();
4897
4898 // Don't generate 24-bit multiplies on values that are in SGPRs, since
4899 // we only have a 32-bit scalar multiply (avoid values being moved to VGPRs
4900 // unnecessarily). isDivergent() is used as an approximation of whether the
4901 // value is in an SGPR.
4902 // This doesn't apply if no s_mul_hi is available (since we'll end up with a
4903 // valu op anyway)
4904 if (!N->isDivergent() && Subtarget->hasSMulHi())
4905 return SDValue();
4906
4907 SelectionDAG &DAG = DCI.DAG;
4908 SDLoc DL(N);
4909
4910 SDValue N0 = N->getOperand(0);
4911 SDValue N1 = N->getOperand(1);
4912
4913 if (!isU24(N0, DAG) || !isU24(N1, DAG))
4914 return SDValue();
4915
4916 N0 = DAG.getZExtOrTrunc(N0, DL, MVT::i32);
4917 N1 = DAG.getZExtOrTrunc(N1, DL, MVT::i32);
4918
4919 SDValue Mulhi = DAG.getNode(AMDGPUISD::MULHI_U24, DL, MVT::i32, N0, N1);
4920 DCI.AddToWorklist(Mulhi.getNode());
4921 return DAG.getZExtOrTrunc(Mulhi, DL, VT);
4922}
4923
4924SDValue AMDGPUTargetLowering::getFFBX_U32(SelectionDAG &DAG,
4925 SDValue Op,
4926 const SDLoc &DL,
4927 unsigned Opc) const {
4928 EVT VT = Op.getValueType();
4929 if (VT.bitsGT(MVT::i32))
4930 return SDValue();
4931
4932 if (VT != MVT::i32)
4933 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i32, Op);
4934
4935 SDValue FFBX = DAG.getNode(Opc, DL, MVT::i32, Op);
4936 if (VT != MVT::i32)
4937 FFBX = DAG.getNode(ISD::TRUNCATE, DL, VT, FFBX);
4938
4939 return FFBX;
4940}
4941
4942// The native instructions return -1 on 0 input. Optimize out a select that
4943// produces -1 on 0.
4944//
4945// TODO: If zero is not undef, we could also do this if the output is compared
4946// against the bitwidth.
4947//
4948// TODO: Should probably combine against FFBH_U32 instead of ctlz directly.
4950 SDValue LHS, SDValue RHS,
4951 DAGCombinerInfo &DCI) const {
4952 if (!isNullConstant(Cond.getOperand(1)))
4953 return SDValue();
4954
4955 SelectionDAG &DAG = DCI.DAG;
4956 ISD::CondCode CCOpcode = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
4957 SDValue CmpLHS = Cond.getOperand(0);
4958
4959 // select (setcc x, 0, eq), -1, (ctlz_zero_poison x) -> ffbh_u32 x
4960 // select (setcc x, 0, eq), -1, (cttz_zero_poison x) -> ffbl_u32 x
4961 if (CCOpcode == ISD::SETEQ &&
4962 (isCtlzOpc(RHS.getOpcode()) || isCttzOpc(RHS.getOpcode())) &&
4963 RHS.getOperand(0) == CmpLHS && isAllOnesConstant(LHS)) {
4964 unsigned Opc =
4965 isCttzOpc(RHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
4966 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
4967 }
4968
4969 // select (setcc x, 0, ne), (ctlz_zero_poison x), -1 -> ffbh_u32 x
4970 // select (setcc x, 0, ne), (cttz_zero_poison x), -1 -> ffbl_u32 x
4971 if (CCOpcode == ISD::SETNE &&
4972 (isCtlzOpc(LHS.getOpcode()) || isCttzOpc(LHS.getOpcode())) &&
4973 LHS.getOperand(0) == CmpLHS && isAllOnesConstant(RHS)) {
4974 unsigned Opc =
4975 isCttzOpc(LHS.getOpcode()) ? AMDGPUISD::FFBL_B32 : AMDGPUISD::FFBH_U32;
4976
4977 return getFFBX_U32(DAG, CmpLHS, SL, Opc);
4978 }
4979
4980 return SDValue();
4981}
4982
4984 unsigned Op,
4985 const SDLoc &SL,
4986 SDValue Cond,
4987 SDValue N1,
4988 SDValue N2) {
4989 SelectionDAG &DAG = DCI.DAG;
4990 EVT VT = N1.getValueType();
4991
4992 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT, Cond,
4993 N1.getOperand(0), N2.getOperand(0));
4994 DCI.AddToWorklist(NewSelect.getNode());
4995 return DAG.getNode(Op, SL, VT, NewSelect);
4996}
4997
4998// Pull a free FP operation out of a select so it may fold into uses.
4999//
5000// select c, (fneg x), (fneg y) -> fneg (select c, x, y)
5001// select c, (fneg x), k -> fneg (select c, x, (fneg k))
5002//
5003// select c, (fabs x), (fabs y) -> fabs (select c, x, y)
5004// select c, (fabs x), +k -> fabs (select c, x, k)
5005SDValue
5007 SDValue N) const {
5008 SelectionDAG &DAG = DCI.DAG;
5009 SDValue Cond = N.getOperand(0);
5010 SDValue LHS = N.getOperand(1);
5011 SDValue RHS = N.getOperand(2);
5012
5013 EVT VT = N.getValueType();
5014 if ((LHS.getOpcode() == ISD::FABS && RHS.getOpcode() == ISD::FABS) ||
5015 (LHS.getOpcode() == ISD::FNEG && RHS.getOpcode() == ISD::FNEG)) {
5017 return SDValue();
5018
5019 return distributeOpThroughSelect(DCI, LHS.getOpcode(),
5020 SDLoc(N), Cond, LHS, RHS);
5021 }
5022
5023 bool Inv = false;
5024 if (RHS.getOpcode() == ISD::FABS || RHS.getOpcode() == ISD::FNEG) {
5025 std::swap(LHS, RHS);
5026 Inv = true;
5027 }
5028
5029 // TODO: Support vector constants.
5031 if ((LHS.getOpcode() == ISD::FNEG || LHS.getOpcode() == ISD::FABS) && CRHS &&
5032 !selectSupportsSourceMods(N.getNode())) {
5033 SDLoc SL(N);
5034 // If one side is an fneg/fabs and the other is a constant, we can push the
5035 // fneg/fabs down. If it's an fabs, the constant needs to be non-negative.
5036 SDValue NewLHS = LHS.getOperand(0);
5037 SDValue NewRHS = RHS;
5038
5039 // Careful: if the neg can be folded up, don't try to pull it back down.
5040 bool ShouldFoldNeg = true;
5041
5042 if (NewLHS.hasOneUse()) {
5043 unsigned Opc = NewLHS.getOpcode();
5044 if (LHS.getOpcode() == ISD::FNEG && fnegFoldsIntoOp(NewLHS.getNode()))
5045 ShouldFoldNeg = false;
5046 if (LHS.getOpcode() == ISD::FABS && Opc == ISD::FMUL)
5047 ShouldFoldNeg = false;
5048 }
5049
5050 if (ShouldFoldNeg) {
5051 if (LHS.getOpcode() == ISD::FABS && CRHS->isNegative())
5052 return SDValue();
5053
5054 // We're going to be forced to use a source modifier anyway, there's no
5055 // point to pulling the negate out unless we can get a size reduction by
5056 // negating the constant.
5057 //
5058 // TODO: Generalize to use getCheaperNegatedExpression which doesn't know
5059 // about cheaper constants.
5060 if (NewLHS.getOpcode() == ISD::FABS &&
5062 return SDValue();
5063
5065 return SDValue();
5066
5067 if (LHS.getOpcode() == ISD::FNEG)
5068 NewRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5069
5070 if (Inv)
5071 std::swap(NewLHS, NewRHS);
5072
5073 SDValue NewSelect = DAG.getNode(ISD::SELECT, SL, VT,
5074 Cond, NewLHS, NewRHS);
5075 DCI.AddToWorklist(NewSelect.getNode());
5076 return DAG.getNode(LHS.getOpcode(), SL, VT, NewSelect);
5077 }
5078 }
5079
5080 return SDValue();
5081}
5082
5084 DAGCombinerInfo &DCI) const {
5085 if (SDValue Folded = foldFreeOpFromSelect(DCI, SDValue(N, 0)))
5086 return Folded;
5087
5088 SDValue Cond = N->getOperand(0);
5089 if (Cond.getOpcode() != ISD::SETCC)
5090 return SDValue();
5091
5092 EVT VT = N->getValueType(0);
5093 SDValue LHS = Cond.getOperand(0);
5094 SDValue RHS = Cond.getOperand(1);
5095 SDValue CC = Cond.getOperand(2);
5096
5097 SDValue True = N->getOperand(1);
5098 SDValue False = N->getOperand(2);
5099
5100 if (Cond.hasOneUse()) { // TODO: Look for multiple select uses.
5101 SelectionDAG &DAG = DCI.DAG;
5102 if (DAG.isConstantValueOfAnyType(True) &&
5103 !DAG.isConstantValueOfAnyType(False)) {
5104 // Swap cmp + select pair to move constant to false input.
5105 // This will allow using VOPC cndmasks more often.
5106 // select (setcc x, y), k, x -> select (setccinv x, y), x, k
5107
5108 SDLoc SL(N);
5109 ISD::CondCode NewCC =
5110 getSetCCInverse(cast<CondCodeSDNode>(CC)->get(), LHS.getValueType());
5111
5112 SDValue NewCond = DAG.getSetCC(SL, Cond.getValueType(), LHS, RHS, NewCC);
5113 return DAG.getNode(ISD::SELECT, SL, VT, NewCond, False, True);
5114 }
5115
5116 if (VT == MVT::f32 && Subtarget->hasFminFmaxLegacy()) {
5118 = combineFMinMaxLegacy(SDLoc(N), VT, LHS, RHS, True, False, CC, DCI);
5119 // Revisit this node so we can catch min3/max3/med3 patterns.
5120 //DCI.AddToWorklist(MinMax.getNode());
5121 return MinMax;
5122 }
5123 }
5124
5125 // There's no reason to not do this if the condition has other uses.
5126 return performCtlz_CttzCombine(SDLoc(N), Cond, True, False, DCI);
5127}
5128
5129static bool isInv2Pi(const APFloat &APF) {
5130 static const APFloat KF16(APFloat::IEEEhalf(), APInt(16, 0x3118));
5131 static const APFloat KF32(APFloat::IEEEsingle(), APInt(32, 0x3e22f983));
5132 static const APFloat KF64(APFloat::IEEEdouble(), APInt(64, 0x3fc45f306dc9c882));
5133
5134 return APF.bitwiseIsEqual(KF16) ||
5135 APF.bitwiseIsEqual(KF32) ||
5136 APF.bitwiseIsEqual(KF64);
5137}
5138
5139// 0 and 1.0 / (0.5 * pi) do not have inline immmediates, so there is an
5140// additional cost to negate them.
5143 if (C->isZero())
5144 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5145
5146 if (Subtarget->hasInv2PiInlineImm() && isInv2Pi(C->getValueAPF()))
5147 return C->isNegative() ? NegatibleCost::Cheaper : NegatibleCost::Expensive;
5148
5150}
5151
5157
5163
5164static unsigned inverseMinMax(unsigned Opc) {
5165 switch (Opc) {
5166 case ISD::FMAXNUM:
5167 return ISD::FMINNUM;
5168 case ISD::FMINNUM:
5169 return ISD::FMAXNUM;
5170 case ISD::FMAXNUM_IEEE:
5171 return ISD::FMINNUM_IEEE;
5172 case ISD::FMINNUM_IEEE:
5173 return ISD::FMAXNUM_IEEE;
5174 case ISD::FMAXIMUM:
5175 return ISD::FMINIMUM;
5176 case ISD::FMINIMUM:
5177 return ISD::FMAXIMUM;
5178 case ISD::FMAXIMUMNUM:
5179 return ISD::FMINIMUMNUM;
5180 case ISD::FMINIMUMNUM:
5181 return ISD::FMAXIMUMNUM;
5182 case AMDGPUISD::FMAX_LEGACY:
5183 return AMDGPUISD::FMIN_LEGACY;
5184 case AMDGPUISD::FMIN_LEGACY:
5185 return AMDGPUISD::FMAX_LEGACY;
5186 default:
5187 llvm_unreachable("invalid min/max opcode");
5188 }
5189}
5190
5191/// \return true if it's profitable to try to push an fneg into its source
5192/// instruction.
5194 // If the input has multiple uses and we can either fold the negate down, or
5195 // the other uses cannot, give up. This both prevents unprofitable
5196 // transformations and infinite loops: we won't repeatedly try to fold around
5197 // a negate that has no 'good' form.
5198 if (N0.hasOneUse()) {
5199 // This may be able to fold into the source, but at a code size cost. Don't
5200 // fold if the fold into the user is free.
5201 if (allUsesHaveSourceMods(N, 0))
5202 return false;
5203 } else {
5204 if (fnegFoldsIntoOp(N0.getNode()) &&
5206 return false;
5207 }
5208
5209 return true;
5210}
5211
5213 DAGCombinerInfo &DCI) const {
5214 SelectionDAG &DAG = DCI.DAG;
5215 SDValue N0 = N->getOperand(0);
5216 EVT VT = N->getValueType(0);
5217
5218 unsigned Opc = N0.getOpcode();
5219
5220 if (!shouldFoldFNegIntoSrc(N, N0))
5221 return SDValue();
5222
5223 SDLoc SL(N);
5224 switch (Opc) {
5225 case ISD::FADD: {
5226 if (!N0->getFlags().hasNoSignedZeros() && !N->getFlags().hasNoSignedZeros())
5227 return SDValue();
5228
5229 // (fneg (fadd x, y)) -> (fadd (fneg x), (fneg y))
5230 SDValue LHS = N0.getOperand(0);
5231 SDValue RHS = N0.getOperand(1);
5232
5233 if (LHS.getOpcode() != ISD::FNEG)
5234 LHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5235 else
5236 LHS = LHS.getOperand(0);
5237
5238 if (RHS.getOpcode() != ISD::FNEG)
5239 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5240 else
5241 RHS = RHS.getOperand(0);
5242
5243 SDValue Res = DAG.getNode(ISD::FADD, SL, VT, LHS, RHS, N0->getFlags());
5244 if (Res.getOpcode() != ISD::FADD)
5245 return SDValue(); // Op got folded away.
5246 if (!N0.hasOneUse())
5247 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5248 return Res;
5249 }
5250 case ISD::FMUL:
5251 case AMDGPUISD::FMUL_LEGACY: {
5252 // (fneg (fmul x, y)) -> (fmul x, (fneg y))
5253 // (fneg (fmul_legacy x, y)) -> (fmul_legacy x, (fneg y))
5254 SDValue LHS = N0.getOperand(0);
5255 SDValue RHS = N0.getOperand(1);
5256
5257 if (LHS.getOpcode() == ISD::FNEG)
5258 LHS = LHS.getOperand(0);
5259 else if (RHS.getOpcode() == ISD::FNEG)
5260 RHS = RHS.getOperand(0);
5261 else
5262 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5263
5264 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, RHS, N0->getFlags());
5265 if (Res.getOpcode() != Opc)
5266 return SDValue(); // Op got folded away.
5267 if (!N0.hasOneUse())
5268 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5269 return Res;
5270 }
5271 case ISD::FMA:
5272 case ISD::FMAD: {
5273 // TODO: handle llvm.amdgcn.fma.legacy
5274 if (!N0->getFlags().hasNoSignedZeros() && !N->getFlags().hasNoSignedZeros())
5275 return SDValue();
5276
5277 // (fneg (fma x, y, z)) -> (fma x, (fneg y), (fneg z))
5278 SDValue LHS = N0.getOperand(0);
5279 SDValue MHS = N0.getOperand(1);
5280 SDValue RHS = N0.getOperand(2);
5281
5282 if (LHS.getOpcode() == ISD::FNEG)
5283 LHS = LHS.getOperand(0);
5284 else if (MHS.getOpcode() == ISD::FNEG)
5285 MHS = MHS.getOperand(0);
5286 else
5287 MHS = DAG.getNode(ISD::FNEG, SL, VT, MHS);
5288
5289 if (RHS.getOpcode() != ISD::FNEG)
5290 RHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5291 else
5292 RHS = RHS.getOperand(0);
5293
5294 SDValue Res = DAG.getNode(Opc, SL, VT, LHS, MHS, RHS);
5295 if (Res.getOpcode() != Opc)
5296 return SDValue(); // Op got folded away.
5297 if (!N0.hasOneUse())
5298 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5299 return Res;
5300 }
5301 case ISD::FMAXNUM:
5302 case ISD::FMINNUM:
5303 case ISD::FMAXNUM_IEEE:
5304 case ISD::FMINNUM_IEEE:
5305 case ISD::FMINIMUM:
5306 case ISD::FMAXIMUM:
5307 case ISD::FMINIMUMNUM:
5308 case ISD::FMAXIMUMNUM:
5309 case AMDGPUISD::FMAX_LEGACY:
5310 case AMDGPUISD::FMIN_LEGACY: {
5311 // fneg (fmaxnum x, y) -> fminnum (fneg x), (fneg y)
5312 // fneg (fminnum x, y) -> fmaxnum (fneg x), (fneg y)
5313 // fneg (fmax_legacy x, y) -> fmin_legacy (fneg x), (fneg y)
5314 // fneg (fmin_legacy x, y) -> fmax_legacy (fneg x), (fneg y)
5315
5316 SDValue LHS = N0.getOperand(0);
5317 SDValue RHS = N0.getOperand(1);
5318
5319 // 0 doesn't have a negated inline immediate.
5320 // TODO: This constant check should be generalized to other operations.
5322 return SDValue();
5323
5324 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, VT, LHS);
5325 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
5326 unsigned Opposite = inverseMinMax(Opc);
5327
5328 SDValue Res = DAG.getNode(Opposite, SL, VT, NegLHS, NegRHS, N0->getFlags());
5329 if (Res.getOpcode() != Opposite)
5330 return SDValue(); // Op got folded away.
5331 if (!N0.hasOneUse())
5332 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Res));
5333 return Res;
5334 }
5335 case AMDGPUISD::FMED3: {
5336 // med3 sorts a NaN input as smaller than everything regardless of its sign,
5337 // so negating all operands does not sign-flip the median when an input may
5338 // be NaN.
5339 if (!N0->getFlags().hasNoNaNs())
5340 return SDValue();
5341
5342 SDValue Ops[3];
5343 for (unsigned I = 0; I < 3; ++I)
5344 Ops[I] = DAG.getNode(ISD::FNEG, SL, VT, N0->getOperand(I), N0->getFlags());
5345
5346 SDValue Res = DAG.getNode(AMDGPUISD::FMED3, SL, VT, Ops, N0->getFlags());
5347 if (Res.getOpcode() != AMDGPUISD::FMED3)
5348 return SDValue(); // Op got folded away.
5349
5350 if (!N0.hasOneUse()) {
5351 SDValue Neg = DAG.getNode(ISD::FNEG, SL, VT, Res);
5352 DAG.ReplaceAllUsesWith(N0, Neg);
5353
5354 for (SDNode *U : Neg->users())
5355 DCI.AddToWorklist(U);
5356 }
5357
5358 return Res;
5359 }
5360 case ISD::FP_EXTEND:
5361 case ISD::FTRUNC:
5362 case ISD::FRINT:
5363 case ISD::FNEARBYINT: // XXX - Should fround be handled?
5364 case ISD::FROUNDEVEN:
5365 case ISD::FSIN:
5366 case ISD::FCANONICALIZE:
5367 case AMDGPUISD::RCP:
5368 case AMDGPUISD::RCP_LEGACY:
5369 case AMDGPUISD::RCP_IFLAG:
5370 case AMDGPUISD::SIN_HW: {
5371 SDValue CvtSrc = N0.getOperand(0);
5372 if (CvtSrc.getOpcode() == ISD::FNEG) {
5373 // (fneg (fp_extend (fneg x))) -> (fp_extend x)
5374 // (fneg (rcp (fneg x))) -> (rcp x)
5375 return DAG.getNode(Opc, SL, VT, CvtSrc.getOperand(0));
5376 }
5377
5378 if (!N0.hasOneUse())
5379 return SDValue();
5380
5381 // (fneg (fp_extend x)) -> (fp_extend (fneg x))
5382 // (fneg (rcp x)) -> (rcp (fneg x))
5383 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5384 return DAG.getNode(Opc, SL, VT, Neg, N0->getFlags());
5385 }
5386 case ISD::FP_ROUND: {
5387 SDValue CvtSrc = N0.getOperand(0);
5388
5389 if (CvtSrc.getOpcode() == ISD::FNEG) {
5390 // (fneg (fp_round (fneg x))) -> (fp_round x)
5391 return DAG.getNode(ISD::FP_ROUND, SL, VT,
5392 CvtSrc.getOperand(0), N0.getOperand(1));
5393 }
5394
5395 if (!N0.hasOneUse())
5396 return SDValue();
5397
5398 // (fneg (fp_round x)) -> (fp_round (fneg x))
5399 SDValue Neg = DAG.getNode(ISD::FNEG, SL, CvtSrc.getValueType(), CvtSrc);
5400 return DAG.getNode(ISD::FP_ROUND, SL, VT, Neg, N0.getOperand(1));
5401 }
5402 case ISD::FP16_TO_FP: {
5403 // v_cvt_f32_f16 supports source modifiers on pre-VI targets without legal
5404 // f16, but legalization of f16 fneg ends up pulling it out of the source.
5405 // Put the fneg back as a legal source operation that can be matched later.
5406 SDLoc SL(N);
5407
5408 SDValue Src = N0.getOperand(0);
5409 EVT SrcVT = Src.getValueType();
5410
5411 // fneg (fp16_to_fp x) -> fp16_to_fp (xor x, 0x8000)
5412 SDValue IntFNeg = DAG.getNode(ISD::XOR, SL, SrcVT, Src,
5413 DAG.getConstant(0x8000, SL, SrcVT));
5414 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFNeg);
5415 }
5416 case ISD::SELECT: {
5417 // fneg (select c, a, b) -> select c, (fneg a), (fneg b)
5418 // TODO: Invert conditions of foldFreeOpFromSelect
5419 return SDValue();
5420 }
5421 case ISD::BITCAST: {
5422 SDLoc SL(N);
5423 SDValue BCSrc = N0.getOperand(0);
5424 if (BCSrc.getOpcode() == ISD::BUILD_VECTOR) {
5425 SDValue HighBits = BCSrc.getOperand(BCSrc.getNumOperands() - 1);
5426 if (VT != MVT::f64 || HighBits.getValueType().getSizeInBits() != 32 ||
5427 !fnegFoldsIntoOp(HighBits.getNode()))
5428 return SDValue();
5429
5430 // f64 fneg only really needs to operate on the high half of of the
5431 // register, so try to force it to an f32 operation to help make use of
5432 // source modifiers.
5433 //
5434 //
5435 // fneg (f64 (bitcast (build_vector x, y))) ->
5436 // f64 (bitcast (build_vector (bitcast i32:x to f32),
5437 // (fneg (bitcast i32:y to f32)))
5438
5439 SDValue CastHi = DAG.getNode(ISD::BITCAST, SL, MVT::f32, HighBits);
5440 SDValue NegHi = DAG.getNode(ISD::FNEG, SL, MVT::f32, CastHi);
5441 SDValue CastBack =
5442 DAG.getNode(ISD::BITCAST, SL, HighBits.getValueType(), NegHi);
5443
5445 Ops.back() = CastBack;
5446 DCI.AddToWorklist(NegHi.getNode());
5447 SDValue Build =
5448 DAG.getNode(ISD::BUILD_VECTOR, SL, BCSrc.getValueType(), Ops);
5449 SDValue Result = DAG.getNode(ISD::BITCAST, SL, VT, Build);
5450
5451 if (!N0.hasOneUse())
5452 DAG.ReplaceAllUsesWith(N0, DAG.getNode(ISD::FNEG, SL, VT, Result));
5453 return Result;
5454 }
5455
5456 if (BCSrc.getOpcode() == ISD::SELECT && VT == MVT::f32 &&
5457 BCSrc.hasOneUse()) {
5458 // fneg (bitcast (f32 (select cond, i32:lhs, i32:rhs))) ->
5459 // select cond, (bitcast i32:lhs to f32), (bitcast i32:rhs to f32)
5460
5461 // TODO: Cast back result for multiple uses is beneficial in some cases.
5462
5463 SDValue LHS =
5464 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(1));
5465 SDValue RHS =
5466 DAG.getNode(ISD::BITCAST, SL, MVT::f32, BCSrc.getOperand(2));
5467
5468 SDValue NegLHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, LHS);
5469 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, MVT::f32, RHS);
5470
5471 return DAG.getNode(ISD::SELECT, SL, MVT::f32, BCSrc.getOperand(0), NegLHS,
5472 NegRHS);
5473 }
5474
5475 return SDValue();
5476 }
5477 default:
5478 return SDValue();
5479 }
5480}
5481
5483 DAGCombinerInfo &DCI) const {
5484 SelectionDAG &DAG = DCI.DAG;
5485 SDValue N0 = N->getOperand(0);
5486
5487 if (!N0.hasOneUse())
5488 return SDValue();
5489
5490 switch (N0.getOpcode()) {
5491 case ISD::FP16_TO_FP: {
5492 assert(!isTypeLegal(MVT::f16) && "should only see if f16 is illegal");
5493 SDLoc SL(N);
5494 SDValue Src = N0.getOperand(0);
5495 EVT SrcVT = Src.getValueType();
5496
5497 // fabs (fp16_to_fp x) -> fp16_to_fp (and x, 0x7fff)
5498 SDValue IntFAbs = DAG.getNode(ISD::AND, SL, SrcVT, Src,
5499 DAG.getConstant(0x7fff, SL, SrcVT));
5500 return DAG.getNode(ISD::FP16_TO_FP, SL, N->getValueType(0), IntFAbs);
5501 }
5502 case ISD::FP_ROUND: {
5503 SDLoc SL(N);
5504 SDValue CvtSrc = N0.getOperand(0);
5505
5506 // fabs (fp_round x) -> fp_round (fabs x)
5507 SDValue Abs = DAG.getNode(ISD::FABS, SL, CvtSrc.getValueType(), CvtSrc,
5508 N->getFlags());
5509 return DAG.getNode(ISD::FP_ROUND, SL, N->getValueType(0), Abs,
5510 N0.getOperand(1), N0->getFlags());
5511 }
5512 default:
5513 return SDValue();
5514 }
5515}
5516
5518 DAGCombinerInfo &DCI) const {
5519 const auto *CFP = dyn_cast<ConstantFPSDNode>(N->getOperand(0));
5520 if (!CFP)
5521 return SDValue();
5522
5523 std::optional<APFloat> Result = AMDGPU::evaluateRcp(CFP->getValueAPF());
5524 if (!Result)
5525 return SDValue();
5526
5527 return DCI.DAG.getConstantFP(*Result, SDLoc(N), N->getValueType(0));
5528}
5529
5531 if (!Subtarget->isGCN())
5532 return false;
5533
5536 auto &ST = DAG.getSubtarget<GCNSubtarget>();
5537 const auto *TII = ST.getInstrInfo();
5538
5539 if (!ST.hasVMovB64Inst() || (!SDConstant && !SDFPConstant))
5540 return false;
5541
5542 if (ST.has64BitLiterals())
5543 return true;
5544
5545 if (SDConstant) {
5546 const APInt &APVal = SDConstant->getAPIntValue();
5547 return isUInt<32>(APVal.getZExtValue()) || TII->isInlineConstant(APVal);
5548 }
5549
5550 APInt Val = SDFPConstant->getValueAPF().bitcastToAPInt();
5551 return isUInt<32>(Val.getZExtValue()) || TII->isInlineConstant(Val);
5552}
5553
5555 DAGCombinerInfo &DCI) const {
5556 SelectionDAG &DAG = DCI.DAG;
5557 SDLoc DL(N);
5558
5559 switch(N->getOpcode()) {
5560 default:
5561 break;
5562 case ISD::BITCAST: {
5563 EVT DestVT = N->getValueType(0);
5564
5565 // Push casts through vector builds. This helps avoid emitting a large
5566 // number of copies when materializing floating point vector constants.
5567 //
5568 // vNt1 bitcast (vNt0 (build_vector t0:x, t0:y)) =>
5569 // vnt1 = build_vector (t1 (bitcast t0:x)), (t1 (bitcast t0:y))
5570 if (DestVT.isVector()) {
5571 SDValue Src = N->getOperand(0);
5572 if (Src.getOpcode() == ISD::BUILD_VECTOR &&
5575 EVT SrcVT = Src.getValueType();
5576 unsigned NElts = DestVT.getVectorNumElements();
5577
5578 if (SrcVT.getVectorNumElements() == NElts) {
5579 EVT DestEltVT = DestVT.getVectorElementType();
5580
5581 SmallVector<SDValue, 8> CastedElts;
5582 SDLoc SL(N);
5583 for (unsigned I = 0, E = SrcVT.getVectorNumElements(); I != E; ++I) {
5584 SDValue Elt = Src.getOperand(I);
5585 CastedElts.push_back(DAG.getNode(ISD::BITCAST, DL, DestEltVT, Elt));
5586 }
5587
5588 return DAG.getBuildVector(DestVT, SL, CastedElts);
5589 }
5590 }
5591 }
5592
5593 if (DestVT.getSizeInBits() != 64 || !DestVT.isVector())
5594 break;
5595
5596 // Fold bitcasts of constants.
5597 //
5598 // v2i32 (bitcast i64:k) -> build_vector lo_32(k), hi_32(k)
5599 // TODO: Generalize and move to DAGCombiner
5600 SDValue Src = N->getOperand(0);
5602 SDLoc SL(N);
5603 if (isInt64ImmLegal(C, DAG))
5604 break;
5605 uint64_t CVal = C->getZExtValue();
5606 SDValue BV = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5607 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5608 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5609 return DAG.getNode(ISD::BITCAST, SL, DestVT, BV);
5610 }
5611
5613 const APInt &Val = C->getValueAPF().bitcastToAPInt();
5614 SDLoc SL(N);
5615 if (isInt64ImmLegal(C, DAG))
5616 break;
5617 uint64_t CVal = Val.getZExtValue();
5618 SDValue Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32,
5619 DAG.getConstant(Lo_32(CVal), SL, MVT::i32),
5620 DAG.getConstant(Hi_32(CVal), SL, MVT::i32));
5621
5622 return DAG.getNode(ISD::BITCAST, SL, DestVT, Vec);
5623 }
5624
5625 break;
5626 }
5627 case ISD::SHL:
5628 case ISD::SRA:
5629 case ISD::SRL: {
5630 // Range metadata can be invalidated when loads are converted to legal types
5631 // (e.g. v2i64 -> v4i32).
5632 // Try to convert vector shl/sra/srl before type legalization so that range
5633 // metadata can be utilized.
5634 if (!(N->getValueType(0).isVector() &&
5637 break;
5638 if (N->getOpcode() == ISD::SHL)
5639 return performShlCombine(N, DCI);
5640 if (N->getOpcode() == ISD::SRA)
5641 return performSraCombine(N, DCI);
5642 return performSrlCombine(N, DCI);
5643 }
5644 case ISD::TRUNCATE:
5645 return performTruncateCombine(N, DCI);
5646 case ISD::MUL:
5647 return performMulCombine(N, DCI);
5648 case AMDGPUISD::MUL_U24:
5649 case AMDGPUISD::MUL_I24: {
5650 if (SDValue Simplified = simplifyMul24(N, DCI))
5651 return Simplified;
5652 break;
5653 }
5654 case AMDGPUISD::MULHI_I24:
5655 case AMDGPUISD::MULHI_U24:
5656 return simplifyMul24(N, DCI);
5657 case ISD::SMUL_LOHI:
5658 case ISD::UMUL_LOHI:
5659 return performMulLoHiCombine(N, DCI);
5660 case ISD::MULHS:
5661 return performMulhsCombine(N, DCI);
5662 case ISD::MULHU:
5663 return performMulhuCombine(N, DCI);
5664 case ISD::SELECT:
5665 return performSelectCombine(N, DCI);
5666 case ISD::FNEG:
5667 return performFNegCombine(N, DCI);
5668 case ISD::FABS:
5669 return performFAbsCombine(N, DCI);
5670 case AMDGPUISD::BFE_I32:
5671 case AMDGPUISD::BFE_U32: {
5672 assert(N->getValueType(0) == MVT::i32 &&
5673 "BFE_I32/BFE_U32 is a 32-bit operation");
5674 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(N->getOperand(2));
5675 if (!Width)
5676 break;
5677
5678 uint32_t WidthVal = Width->getZExtValue() & 0x1f;
5679 if (WidthVal == 0)
5680 return DAG.getConstant(0, DL, MVT::i32);
5681
5683 if (!Offset)
5684 break;
5685
5686 SDValue BitsFrom = N->getOperand(0);
5687 uint32_t OffsetVal = Offset->getZExtValue() & 0x1f;
5688
5689 bool Signed = N->getOpcode() == AMDGPUISD::BFE_I32;
5690
5691 if (OffsetVal == 0) {
5692 // This is already sign / zero extended, so try to fold away extra BFEs.
5693 EVT SmallVT = EVT::getIntegerVT(*DAG.getContext(), WidthVal);
5694 if (Signed) {
5695 if (DAG.ComputeNumSignBits(BitsFrom) >= 32 - WidthVal + 1)
5696 return BitsFrom;
5697
5698 // This is a sign_extend_inreg. Replace it to take advantage of existing
5699 // DAG Combines. If not eliminated, we will match back to BFE during
5700 // selection.
5701
5702 // TODO: The sext_inreg of extended types ends, although we can could
5703 // handle them in a single BFE.
5704 return DAG.getNode(ISD::SIGN_EXTEND_INREG, DL, MVT::i32, BitsFrom,
5705 DAG.getValueType(SmallVT));
5706 }
5707
5708 if (DAG.MaskedValueIsZero(BitsFrom,
5709 APInt::getHighBitsSet(32, 32 - WidthVal)))
5710 return BitsFrom;
5711
5712 return DAG.getZeroExtendInReg(BitsFrom, DL, SmallVT);
5713 }
5714
5715 if (ConstantSDNode *CVal = dyn_cast<ConstantSDNode>(BitsFrom)) {
5716 if (Signed) {
5717 return constantFoldBFE<int32_t>(DAG,
5718 CVal->getSExtValue(),
5719 OffsetVal,
5720 WidthVal,
5721 DL);
5722 }
5723
5724 return constantFoldBFE<uint32_t>(DAG,
5725 CVal->getZExtValue(),
5726 OffsetVal,
5727 WidthVal,
5728 DL);
5729 }
5730
5731 if ((OffsetVal + WidthVal) >= 32 &&
5732 !(OffsetVal == 16 && WidthVal == 16 && Subtarget->hasSDWA())) {
5733 SDValue ShiftVal = DAG.getConstant(OffsetVal, DL, MVT::i32);
5734 return DAG.getNode(Signed ? ISD::SRA : ISD::SRL, DL, MVT::i32,
5735 BitsFrom, ShiftVal);
5736 }
5737
5738 if (BitsFrom.hasOneUse()) {
5739 APInt Demanded = APInt::getBitsSet(32,
5740 OffsetVal,
5741 OffsetVal + WidthVal);
5742
5745 !DCI.isBeforeLegalizeOps());
5746 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
5747 if (TLI.ShrinkDemandedConstant(BitsFrom, Demanded, TLO) ||
5748 TLI.SimplifyDemandedBits(BitsFrom, Demanded, Known, TLO)) {
5749 DCI.CommitTargetLoweringOpt(TLO);
5750 }
5751 }
5752
5753 break;
5754 }
5755 case ISD::LOAD:
5756 return performLoadCombine(N, DCI);
5757 case ISD::STORE:
5758 return performStoreCombine(N, DCI);
5759 case AMDGPUISD::RCP:
5760 case AMDGPUISD::RCP_IFLAG:
5761 return performRcpCombine(N, DCI);
5762 case ISD::AssertZext:
5763 case ISD::AssertSext:
5764 return performAssertSZExtCombine(N, DCI);
5766 return performIntrinsicWOChainCombine(N, DCI);
5767 case AMDGPUISD::FMAD_FTZ: {
5768 SDValue N0 = N->getOperand(0);
5769 SDValue N1 = N->getOperand(1);
5770 SDValue N2 = N->getOperand(2);
5771 EVT VT = N->getValueType(0);
5772
5773 // FMAD_FTZ is a FMAD + flush denormals to zero.
5774 // We flush the inputs, the intermediate step, and the output.
5778 if (N0CFP && N1CFP && N2CFP) {
5779 const auto FTZ = [](const APFloat &V) {
5780 if (V.isDenormal()) {
5781 APFloat Zero(V.getSemantics(), 0);
5782 return V.isNegative() ? -Zero : Zero;
5783 }
5784 return V;
5785 };
5786
5787 APFloat V0 = FTZ(N0CFP->getValueAPF());
5788 APFloat V1 = FTZ(N1CFP->getValueAPF());
5789 APFloat V2 = FTZ(N2CFP->getValueAPF());
5791 V0 = FTZ(V0);
5793 return DAG.getConstantFP(FTZ(V0), DL, VT);
5794 }
5795 break;
5796 }
5797 }
5798 return SDValue();
5799}
5800
5802 SDValue Op, const APInt &OriginalDemandedBits,
5803 const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO,
5804 unsigned Depth) const {
5805 switch (Op.getOpcode()) {
5807 switch (Op.getConstantOperandVal(0)) {
5808 case Intrinsic::amdgcn_readfirstlane:
5809 case Intrinsic::amdgcn_readlane:
5810 case Intrinsic::amdgcn_wwm: {
5811 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5812 OriginalDemandedElts, Known, TLO, Depth + 1))
5813 return true;
5814 break;
5815 }
5816 case Intrinsic::amdgcn_set_inactive:
5817 case Intrinsic::amdgcn_set_inactive_chain_arg: {
5818 // The result is operand 1 in active lanes and operand 2 in inactive
5819 // lanes, so the known bits are the intersection of both operands.
5820 KnownBits KnownValue, KnownInactive;
5821 if (SimplifyDemandedBits(Op.getOperand(1), OriginalDemandedBits,
5822 OriginalDemandedElts, KnownValue, TLO,
5823 Depth + 1))
5824 return true;
5825 if (SimplifyDemandedBits(Op.getOperand(2), OriginalDemandedBits,
5826 OriginalDemandedElts, KnownInactive, TLO,
5827 Depth + 1))
5828 return true;
5829 Known = KnownValue.intersectWith(KnownInactive);
5830 break;
5831 }
5832 default:
5833 break;
5834 }
5835 break;
5836 }
5837 default:
5838 break;
5839 }
5840
5841 return false;
5842}
5843
5844//===----------------------------------------------------------------------===//
5845// Helper functions
5846//===----------------------------------------------------------------------===//
5847
5849 const TargetRegisterClass *RC,
5850 Register Reg, EVT VT,
5851 const SDLoc &SL,
5852 bool RawReg) const {
5854 MachineRegisterInfo &MRI = MF.getRegInfo();
5855 Register VReg;
5856
5857 if (!MRI.isLiveIn(Reg)) {
5858 VReg = MRI.createVirtualRegister(RC);
5859 MRI.addLiveIn(Reg, VReg);
5860 } else {
5861 VReg = MRI.getLiveInVirtReg(Reg);
5862 }
5863
5864 if (RawReg)
5865 return DAG.getRegister(VReg, VT);
5866
5867 return DAG.getCopyFromReg(DAG.getEntryNode(), SL, VReg, VT);
5868}
5869
5870// This may be called multiple times, and nothing prevents creating multiple
5871// objects at the same offset. See if we already defined this object.
5873 int64_t Offset) {
5874 for (int I = MFI.getObjectIndexBegin(); I < 0; ++I) {
5875 if (MFI.getObjectOffset(I) == Offset) {
5876 assert(MFI.getObjectSize(I) == Size);
5877 return I;
5878 }
5879 }
5880
5881 return MFI.CreateFixedObject(Size, Offset, true);
5882}
5883
5885 EVT VT,
5886 const SDLoc &SL,
5887 int64_t Offset) const {
5889 MachineFrameInfo &MFI = MF.getFrameInfo();
5890 int FI = getOrCreateFixedStackObject(MFI, VT.getStoreSize(), Offset);
5891
5892 auto SrcPtrInfo = MachinePointerInfo::getStack(MF, Offset);
5893 SDValue Ptr = DAG.getFrameIndex(FI, MVT::i32);
5894
5895 return DAG.getLoad(VT, SL, DAG.getEntryNode(), Ptr, SrcPtrInfo, Align(4),
5898}
5899
5901 const SDLoc &SL,
5902 SDValue Chain,
5903 SDValue ArgVal,
5904 int64_t Offset) const {
5908
5909 SDValue Ptr = DAG.getConstant(Offset, SL, MVT::i32);
5910 // Stores to the argument stack area are relative to the stack pointer.
5911 SDValue SP =
5912 DAG.getCopyFromReg(Chain, SL, Info->getStackPtrOffsetReg(), MVT::i32);
5913 Ptr = DAG.getNode(ISD::ADD, SL, MVT::i32, SP, Ptr);
5914 SDValue Store = DAG.getStore(Chain, SL, ArgVal, Ptr, DstInfo, Align(4),
5916 return Store;
5917}
5918
5920 const TargetRegisterClass *RC,
5921 EVT VT, const SDLoc &SL,
5922 const ArgDescriptor &Arg) const {
5923 assert(Arg && "Attempting to load missing argument");
5924
5925 SDValue V = Arg.isRegister() ?
5926 CreateLiveInRegister(DAG, RC, Arg.getRegister(), VT, SL) :
5927 loadStackInputValue(DAG, VT, SL, Arg.getStackOffset());
5928
5929 if (!Arg.isMasked())
5930 return V;
5931
5932 unsigned Mask = Arg.getMask();
5933 unsigned Shift = llvm::countr_zero<unsigned>(Mask);
5934 V = DAG.getNode(ISD::SRL, SL, VT, V,
5935 DAG.getShiftAmountConstant(Shift, VT, SL));
5936 return DAG.getNode(ISD::AND, SL, VT, V,
5937 DAG.getConstant(Mask >> Shift, SL, VT));
5938}
5939
5941 uint64_t ExplicitKernArgSize, const ImplicitParameter Param) const {
5942 unsigned ExplicitArgOffset = Subtarget->getExplicitKernelArgOffset();
5943 const Align Alignment = Subtarget->getAlignmentForImplicitArgPtr();
5944 uint64_t ArgOffset =
5945 alignTo(ExplicitKernArgSize, Alignment) + ExplicitArgOffset;
5946 switch (Param) {
5947 case FIRST_IMPLICIT:
5948 return ArgOffset;
5949 case PRIVATE_BASE:
5951 case SHARED_BASE:
5952 return ArgOffset + AMDGPU::ImplicitArg::SHARED_BASE_OFFSET;
5953 case QUEUE_PTR:
5954 return ArgOffset + AMDGPU::ImplicitArg::QUEUE_PTR_OFFSET;
5955 }
5956 llvm_unreachable("unexpected implicit parameter type");
5957}
5958
5965
5967 SelectionDAG &DAG, int Enabled,
5968 int &RefinementSteps,
5969 bool &UseOneConstNR,
5970 bool Reciprocal) const {
5971 EVT VT = Operand.getValueType();
5972
5973 if (VT == MVT::f32) {
5974 RefinementSteps = 0;
5975 return DAG.getNode(AMDGPUISD::RSQ, SDLoc(Operand), VT, Operand);
5976 }
5977
5978 // TODO: There is also f64 rsq instruction, but the documentation is less
5979 // clear on its precision.
5980
5981 return SDValue();
5982}
5983
5985 SelectionDAG &DAG, int Enabled,
5986 int &RefinementSteps) const {
5987 EVT VT = Operand.getValueType();
5988
5989 if (VT == MVT::f32) {
5990 // Reciprocal, < 1 ulp error.
5991 //
5992 // This reciprocal approximation converges to < 0.5 ulp error with one
5993 // newton rhapson performed with two fused multiple adds (FMAs).
5994
5995 RefinementSteps = 0;
5996 return DAG.getNode(AMDGPUISD::RCP, SDLoc(Operand), VT, Operand);
5997 }
5998
5999 // TODO: There is also f64 rcp instruction, but the documentation is less
6000 // clear on its precision.
6001
6002 return SDValue();
6003}
6004
6005static unsigned workitemIntrinsicDim(unsigned ID) {
6006 switch (ID) {
6007 case Intrinsic::amdgcn_workitem_id_x:
6008 return 0;
6009 case Intrinsic::amdgcn_workitem_id_y:
6010 return 1;
6011 case Intrinsic::amdgcn_workitem_id_z:
6012 return 2;
6013 default:
6014 llvm_unreachable("not a workitem intrinsic");
6015 }
6016}
6017
6019 const SDValue Op, KnownBits &Known,
6020 const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const {
6021
6022 Known.resetAll(); // Don't know anything.
6023
6024 unsigned Opc = Op.getOpcode();
6025
6026 switch (Opc) {
6027 default:
6028 break;
6029 case AMDGPUISD::CARRY:
6030 case AMDGPUISD::BORROW: {
6031 Known.Zero = APInt::getHighBitsSet(32, 31);
6032 break;
6033 }
6034
6035 case AMDGPUISD::BFE_I32:
6036 case AMDGPUISD::BFE_U32: {
6037 ConstantSDNode *CWidth = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6038 if (!CWidth)
6039 return;
6040
6041 uint32_t Width = CWidth->getZExtValue() & 0x1f;
6042
6043 if (Opc == AMDGPUISD::BFE_U32)
6044 Known.Zero = APInt::getHighBitsSet(32, 32 - Width);
6045
6046 break;
6047 }
6048 case AMDGPUISD::FP_TO_FP16: {
6049 unsigned BitWidth = Known.getBitWidth();
6050
6051 // High bits are zero.
6053 break;
6054 }
6055 case AMDGPUISD::MUL_U24:
6056 case AMDGPUISD::MUL_I24: {
6057 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6058 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6059 unsigned BitWidth = Op.getScalarValueSizeInBits();
6060
6061 // Sign/Zero extend from 24 bits.
6062 if (Opc == AMDGPUISD::MUL_I24) {
6063 LHSKnown = LHSKnown.trunc(24).sext(BitWidth);
6064 RHSKnown = RHSKnown.trunc(24).sext(BitWidth);
6065 } else {
6066 LHSKnown = LHSKnown.trunc(24).zext(BitWidth);
6067 RHSKnown = RHSKnown.trunc(24).zext(BitWidth);
6068 }
6069
6070 // TODO: SelfMultiply can be poison, but not undef.
6071 bool SelfMultiply = Op.getOperand(0) == Op.getOperand(1);
6072 if (SelfMultiply)
6073 SelfMultiply &= DAG.isGuaranteedNotToBeUndefOrPoison(
6074 Op.getOperand(0), DemandedElts, UndefPoisonKind::UndefOrPoison,
6075 Depth + 1);
6076
6077 Known = KnownBits::mul(LHSKnown, RHSKnown, SelfMultiply);
6078 break;
6079 }
6080 case AMDGPUISD::PERM: {
6081 ConstantSDNode *CMask = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6082 if (!CMask)
6083 return;
6084
6085 KnownBits LHSKnown = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6086 KnownBits RHSKnown = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6087 unsigned Sel = CMask->getZExtValue();
6088
6089 for (unsigned I = 0; I < 32; I += 8) {
6090 unsigned SelBits = Sel & 0xff;
6091 if (SelBits < 4) {
6092 SelBits *= 8;
6093 Known.One |= ((RHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6094 Known.Zero |= ((RHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6095 } else if (SelBits < 7) {
6096 SelBits = (SelBits & 3) * 8;
6097 Known.One |= ((LHSKnown.One.getZExtValue() >> SelBits) & 0xff) << I;
6098 Known.Zero |= ((LHSKnown.Zero.getZExtValue() >> SelBits) & 0xff) << I;
6099 } else if (SelBits == 0x0c) {
6100 Known.Zero |= 0xFFull << I;
6101 } else if (SelBits > 0x0c) {
6102 Known.One |= 0xFFull << I;
6103 }
6104 Sel >>= 8;
6105 }
6106 break;
6107 }
6108 case AMDGPUISD::BUFFER_LOAD_UBYTE: {
6109 Known.Zero.setHighBits(24);
6110 break;
6111 }
6112 case AMDGPUISD::BUFFER_LOAD_USHORT: {
6113 Known.Zero.setHighBits(16);
6114 break;
6115 }
6116 case AMDGPUISD::LDS: {
6117 auto *GA = cast<GlobalAddressSDNode>(Op.getOperand(0).getNode());
6118 Align Alignment = GA->getGlobal()->getPointerAlignment(DAG.getDataLayout());
6119
6120 Known.Zero.setHighBits(16);
6121 Known.Zero.setLowBits(Log2(Alignment));
6122 break;
6123 }
6124 case AMDGPUISD::SMIN3:
6125 case AMDGPUISD::SMAX3:
6126 case AMDGPUISD::SMED3:
6127 case AMDGPUISD::UMIN3:
6128 case AMDGPUISD::UMAX3:
6129 case AMDGPUISD::UMED3: {
6130 KnownBits Known2 = DAG.computeKnownBits(Op.getOperand(2), Depth + 1);
6131 if (Known2.isUnknown())
6132 break;
6133
6134 KnownBits Known1 = DAG.computeKnownBits(Op.getOperand(1), Depth + 1);
6135 if (Known1.isUnknown())
6136 break;
6137
6138 KnownBits Known0 = DAG.computeKnownBits(Op.getOperand(0), Depth + 1);
6139 if (Known0.isUnknown())
6140 break;
6141
6142 // TODO: Handle LeadZero/LeadOne from UMIN/UMAX handling.
6143 Known.Zero = Known0.Zero & Known1.Zero & Known2.Zero;
6144 Known.One = Known0.One & Known1.One & Known2.One;
6145 break;
6146 }
6148 unsigned IID = Op.getConstantOperandVal(0);
6149 switch (IID) {
6150 case Intrinsic::amdgcn_workitem_id_x:
6151 case Intrinsic::amdgcn_workitem_id_y:
6152 case Intrinsic::amdgcn_workitem_id_z: {
6153 unsigned MaxValue = Subtarget->getMaxWorkitemID(
6155 Known.Zero.setHighBits(llvm::countl_zero(MaxValue));
6156 break;
6157 }
6158 default:
6159 break;
6160 }
6161 }
6162 }
6163}
6164
6166 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6167 unsigned Depth) const {
6168 switch (Op.getOpcode()) {
6169 case AMDGPUISD::BFE_I32: {
6170 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6171 if (!Width)
6172 return 1;
6173
6174 unsigned SignBits = 32 - (Width->getZExtValue() & 0x1f) + 1;
6175 if (!isNullConstant(Op.getOperand(1)))
6176 return SignBits;
6177
6178 // TODO: Could probably figure something out with non-0 offsets.
6179 unsigned Op0SignBits = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6180 return std::max(SignBits, Op0SignBits);
6181 }
6182
6183 case AMDGPUISD::BFE_U32: {
6184 ConstantSDNode *Width = dyn_cast<ConstantSDNode>(Op.getOperand(2));
6185 return Width ? 32 - (Width->getZExtValue() & 0x1f) : 1;
6186 }
6187
6188 case AMDGPUISD::CARRY:
6189 case AMDGPUISD::BORROW:
6190 return 31;
6191 case AMDGPUISD::BUFFER_LOAD_BYTE:
6192 return 25;
6193 case AMDGPUISD::BUFFER_LOAD_SHORT:
6194 return 17;
6195 case AMDGPUISD::BUFFER_LOAD_UBYTE:
6196 return 24;
6197 case AMDGPUISD::BUFFER_LOAD_USHORT:
6198 return 16;
6199 case AMDGPUISD::FP_TO_FP16:
6200 return 16;
6201 case AMDGPUISD::SMIN3:
6202 case AMDGPUISD::SMAX3:
6203 case AMDGPUISD::SMED3:
6204 case AMDGPUISD::UMIN3:
6205 case AMDGPUISD::UMAX3:
6206 case AMDGPUISD::UMED3: {
6207 unsigned Tmp2 = DAG.ComputeNumSignBits(Op.getOperand(2), Depth + 1);
6208 if (Tmp2 == 1)
6209 return 1; // Early out.
6210
6211 unsigned Tmp1 = DAG.ComputeNumSignBits(Op.getOperand(1), Depth + 1);
6212 if (Tmp1 == 1)
6213 return 1; // Early out.
6214
6215 unsigned Tmp0 = DAG.ComputeNumSignBits(Op.getOperand(0), Depth + 1);
6216 if (Tmp0 == 1)
6217 return 1; // Early out.
6218
6219 return std::min({Tmp0, Tmp1, Tmp2});
6220 }
6221 default:
6222 return 1;
6223 }
6224}
6225
6227 GISelValueTracking &Analysis, Register R, const APInt &DemandedElts,
6228 const MachineRegisterInfo &MRI, unsigned Depth) const {
6229 const MachineInstr *MI = MRI.getVRegDef(R);
6230 if (!MI)
6231 return 1;
6232
6233 // TODO: Check range metadata on MMO.
6234 switch (MI->getOpcode()) {
6235 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SBYTE:
6236 return 25;
6237 case AMDGPU::G_AMDGPU_BUFFER_LOAD_SSHORT:
6238 return 17;
6239 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
6240 return 24;
6241 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
6242 return 16;
6243 case AMDGPU::G_AMDGPU_SMED3:
6244 case AMDGPU::G_AMDGPU_UMED3: {
6245 auto [Dst, Src0, Src1, Src2] = MI->getFirst4Regs();
6246 unsigned Tmp2 = Analysis.computeNumSignBits(Src2, DemandedElts, Depth + 1);
6247 if (Tmp2 == 1)
6248 return 1;
6249 unsigned Tmp1 = Analysis.computeNumSignBits(Src1, DemandedElts, Depth + 1);
6250 if (Tmp1 == 1)
6251 return 1;
6252 unsigned Tmp0 = Analysis.computeNumSignBits(Src0, DemandedElts, Depth + 1);
6253 if (Tmp0 == 1)
6254 return 1;
6255 return std::min({Tmp0, Tmp1, Tmp2});
6256 }
6257 default:
6258 return 1;
6259 }
6260}
6261
6263 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
6264 UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const {
6265 unsigned Opcode = Op.getOpcode();
6266 switch (Opcode) {
6267 case AMDGPUISD::BFE_I32:
6268 case AMDGPUISD::BFE_U32:
6269 return false;
6270 }
6272 Op, DemandedElts, DAG, Kind, ConsiderFlags, Depth);
6273}
6274
6276 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN,
6277 unsigned Depth) const {
6278 unsigned Opcode = Op.getOpcode();
6279 switch (Opcode) {
6280 case AMDGPUISD::FMIN_LEGACY:
6281 case AMDGPUISD::FMAX_LEGACY:
6282 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6283 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6284 case AMDGPUISD::FMUL_LEGACY:
6285 case AMDGPUISD::CVT_PKRTZ_F16_F32: {
6286 if (SNaN)
6287 return true;
6288 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6289 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6290 }
6291 case AMDGPUISD::FMED3:
6292 case AMDGPUISD::FMIN3:
6293 case AMDGPUISD::FMAX3:
6294 case AMDGPUISD::FMINIMUM3:
6295 case AMDGPUISD::FMAXIMUM3:
6296 case AMDGPUISD::FMAD_FTZ: {
6297 if (SNaN)
6298 return true;
6299 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1) &&
6300 DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6301 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6302 }
6303 case AMDGPUISD::CVT_F32_UBYTE0:
6304 case AMDGPUISD::CVT_F32_UBYTE1:
6305 case AMDGPUISD::CVT_F32_UBYTE2:
6306 case AMDGPUISD::CVT_F32_UBYTE3:
6307 return true;
6308
6309 case AMDGPUISD::RCP:
6310 case AMDGPUISD::RSQ:
6311 case AMDGPUISD::RCP_LEGACY:
6312 case AMDGPUISD::RSQ_CLAMP: {
6313 if (SNaN)
6314 return true;
6315
6316 // TODO: Need is known positive check.
6317 return false;
6318 }
6319 case ISD::FLDEXP:
6320 case AMDGPUISD::FRACT: {
6321 if (SNaN)
6322 return true;
6323 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1);
6324 }
6325 case AMDGPUISD::DIV_SCALE:
6326 case AMDGPUISD::DIV_FMAS:
6327 case AMDGPUISD::DIV_FIXUP:
6328 // TODO: Refine on operands.
6329 return SNaN;
6330 case AMDGPUISD::SIN_HW:
6331 case AMDGPUISD::COS_HW: {
6332 // TODO: Need check for infinity
6333 return SNaN;
6334 }
6336 unsigned IntrinsicID = Op.getConstantOperandVal(0);
6337 // TODO: Handle more intrinsics
6338 switch (IntrinsicID) {
6339 case Intrinsic::amdgcn_cubeid:
6340 case Intrinsic::amdgcn_cvt_off_f32_i4:
6341 return true;
6342
6343 case Intrinsic::amdgcn_frexp_mant: {
6344 if (SNaN)
6345 return true;
6346 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1);
6347 }
6348 case Intrinsic::amdgcn_cvt_pkrtz: {
6349 if (SNaN)
6350 return true;
6351 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6352 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1);
6353 }
6354 case Intrinsic::amdgcn_rcp:
6355 case Intrinsic::amdgcn_rsq:
6356 case Intrinsic::amdgcn_rcp_legacy:
6357 case Intrinsic::amdgcn_rsq_legacy:
6358 case Intrinsic::amdgcn_rsq_clamp:
6359 case Intrinsic::amdgcn_tanh: {
6360 if (SNaN)
6361 return true;
6362
6363 // TODO: Need is known positive check.
6364 return false;
6365 }
6366 case Intrinsic::amdgcn_trig_preop:
6367 case Intrinsic::amdgcn_fdot2:
6368 // TODO: Refine on operand
6369 return SNaN;
6370 case Intrinsic::amdgcn_fma_legacy:
6371 if (SNaN)
6372 return true;
6373 return DAG.isKnownNeverNaN(Op.getOperand(1), SNaN, Depth + 1) &&
6374 DAG.isKnownNeverNaN(Op.getOperand(2), SNaN, Depth + 1) &&
6375 DAG.isKnownNeverNaN(Op.getOperand(3), SNaN, Depth + 1);
6376 default:
6377 return false;
6378 }
6379 }
6380 default:
6381 return false;
6382 }
6383}
6384
6386 Register N0, Register N1) const {
6387 return MRI.hasOneNonDBGUse(N0); // FIXME: handle regbanks
6388}
return SDValue()
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static LLVM_READONLY bool hasSourceMods(const MachineInstr &MI)
static bool isInv2Pi(const APFloat &APF)
static LLVM_READONLY bool opMustUseVOP3Encoding(const MachineInstr &MI, const MachineRegisterInfo &MRI)
returns true if the operation will definitely need to use a 64-bit encoding, and thus will use a VOP3...
static unsigned inverseMinMax(unsigned Opc)
unsigned Imm
static SDValue extractF64Exponent(SDValue Hi, const SDLoc &SL, SelectionDAG &DAG)
static unsigned workitemIntrinsicDim(unsigned ID)
static int getOrCreateFixedStackObject(MachineFrameInfo &MFI, unsigned Size, int64_t Offset)
static SDValue constantFoldBFE(SelectionDAG &DAG, IntTy Src0, uint32_t Offset, uint32_t Width, const SDLoc &DL)
static SDValue getMad(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue X, SDValue Y, SDValue C, SDNodeFlags Flags=SDNodeFlags())
static SDValue getAddOneOp(const SDNode *V)
If V is an add of a constant 1, returns the other operand.
static LLVM_READONLY bool selectSupportsSourceMods(const SDNode *N)
Return true if v_cndmask_b32 will support fabs/fneg source modifiers for the type for ISD::SELECT.
static cl::opt< bool > AMDGPUBypassSlowDiv("amdgpu-bypass-slow-div", cl::desc("Skip 64-bit divide for dynamic 32-bit values"), cl::init(true))
static SDValue getMul24(SelectionDAG &DAG, const SDLoc &SL, SDValue N0, SDValue N1, unsigned Size, bool Signed)
static bool fnegFoldsIntoOp(const SDNode *N)
static bool isI24(SDValue Op, SelectionDAG &DAG)
static bool isCttzOpc(unsigned Opc)
static bool isU24(SDValue Op, SelectionDAG &DAG)
static SDValue peekFPSignOps(SDValue Val)
static bool valueIsKnownNeverF32Denorm(SDValue Src)
Return true if it's known that Src can never be an f32 denormal value.
static SDValue distributeOpThroughSelect(TargetLowering::DAGCombinerInfo &DCI, unsigned Op, const SDLoc &SL, SDValue Cond, SDValue N1, SDValue N2)
static SDValue peekFNeg(SDValue Val)
static SDValue simplifyMul24(SDNode *Node24, TargetLowering::DAGCombinerInfo &DCI)
static bool isCtlzOpc(unsigned Opc)
static LLVM_READNONE bool fnegFoldsIntoOpcode(unsigned Opc)
static bool hasVolatileUser(SDNode *Val)
Interface definition of the TargetLowering class that is common to all AMD GPUs.
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
block Block Frequency Analysis
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< StatepointGC > D("statepoint-example", "an example strategy for statepoint")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
#define LLVM_READNONE
Definition Compiler.h:323
#define LLVM_READONLY
Definition Compiler.h:330
Provides analysis for querying information about KnownBits during GISel passes.
const HexagonInstrInfo * TII
static MaybeAlign getAlign(Value *Ptr)
IRTranslator LLVM IR MI
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
#define T
#define P(N)
const SmallVectorImpl< MachineOperand > & Cond
#define CH(x, y, z)
Definition SHA256.cpp:34
Func MI getDebugLoc()))
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
BinaryOperator * Mul
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
unsigned allocateLDSGlobal(const DataLayout &DL, const GlobalVariable &GV)
void recordNumNamedBarriers(uint32_t GVAddr, unsigned BarCnt)
static std::optional< uint32_t > getLDSAbsoluteAddress(const GlobalValue &GV)
static unsigned numBitsSigned(SDValue Op, SelectionDAG &DAG)
SDValue combineFMinMaxLegacy(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, DAGCombinerInfo &DCI) const
Generate Min/Max node.
unsigned ComputeNumSignBitsForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue performMulhuCombine(SDNode *N, DAGCombinerInfo &DCI) const
EVT getTypeForExtReturn(LLVMContext &Context, EVT VT, ISD::NodeType ExtendKind) const override
Return the type that should be used to zero or sign extend a zeroext/signext integer return value.
SDValue SplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Split a vector load into 2 loads of half the vector.
SDValue LowerCONCAT_VECTORS(SDValue Op, SelectionDAG &DAG) const
SDValue performLoadCombine(SDNode *N, DAGCombinerInfo &DCI) const
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
SDValue LowerF64ToF16Safe(SDValue Src, const SDLoc &DL, SelectionDAG &DAG) const
SDValue LowerFROUND(SDValue Op, SelectionDAG &DAG) const
SDValue storeStackInputValue(SelectionDAG &DAG, const SDLoc &SL, SDValue Chain, SDValue ArgVal, int64_t Offset) const
bool storeOfVectorConstantIsCheap(bool IsZero, EVT MemVT, unsigned NumElem, unsigned AS) const override
Return true if it is expected to be cheaper to do a store of vector constant with the given size and ...
SDValue LowerEXTRACT_SUBVECTOR(SDValue Op, SelectionDAG &DAG) const
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool shouldCombineMemoryType(EVT VT) const
SDValue splitBinaryBitConstantOpImpl(DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS, uint32_t ValLo, uint32_t ValHi) const
Split the 64-bit value LHS into two 32-bit components, and perform the binary operation Opc to it wit...
SDValue lowerUnhandledCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals, StringRef Reason) const
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue performAssertSZExtCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isTruncateFree(EVT Src, EVT Dest) const override
bool aggressivelyPreferBuildVectorSources(EVT VecVT) const override
SDValue LowerFCEIL(SDValue Op, SelectionDAG &DAG) const
TargetLowering::NegatibleCost getConstantNegateCost(const ConstantFPSDNode *C) const
SDValue LowerFLOGUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, bool IsLog10, SDNodeFlags Flags) const
SDValue performMulhsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXPUnsafeImpl(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags, bool IsExp10) const
bool isSDNodeAlwaysUniform(const SDNode *N) const override
bool isDesirableToCommuteWithShift(const SDNode *N, CombineLevel Level) const override
Return true if it is profitable to move this shift by a constant amount through its operand,...
SDValue performShlCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isCheapToSpeculateCtlz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic ctlz.
SDValue LowerSDIVREM(SDValue Op, SelectionDAG &DAG) const
bool isFNegFree(EVT VT) const override
Return true if an fneg operation is free to the point where it is never worthwhile to replace it with...
SDValue LowerFLOG10(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP64(SDValue Op, SelectionDAG &DAG, bool Signed) const
unsigned computeNumSignBitsForTargetInstr(GISelValueTracking &Analysis, Register R, const APInt &DemandedElts, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
This method can be implemented by targets that want to expose additional information about sign bits ...
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue LowerFP_TO_FP16(SDValue Op, SelectionDAG &DAG) const
SDValue addTokenForArgument(SDValue Chain, SelectionDAG &DAG, MachineFrameInfo &MFI, int ClobberedFI) const
bool isConstantCheaperToNegate(SDValue N) const
bool isReassocProfitable(MachineRegisterInfo &MRI, Register N0, Register N1) const override
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
static bool needsDenormHandlingF32(const SelectionDAG &DAG, SDValue Src, SDNodeFlags Flags)
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue lowerFEXPF64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFFLOOR(SDValue Op, SelectionDAG &DAG) const
SDValue performSelectCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue performFNegCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const
bool isConstantCostlierToNegate(SDValue N) const
SDValue loadInputValue(SelectionDAG &DAG, const TargetRegisterClass *RC, EVT VT, const SDLoc &SL, const ArgDescriptor &Arg) const
SDValue lowerFEXP10Unsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
Emit approx-funcs appropriate lowering for exp10.
bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtType, EVT ExtVT, std::optional< unsigned > ByteOffset) const override
Return true if it is profitable to reduce a load to a smaller type.
SDValue LowerUINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const override
Return true if Op can create undef or poison from non-undef & non-poison operands.
bool isCheapToSpeculateCttz(Type *Ty) const override
Return true if it is cheap to speculate a call to intrinsic cttz.
SDValue performCtlz_CttzCombine(const SDLoc &SL, SDValue Cond, SDValue LHS, SDValue RHS, DAGCombinerInfo &DCI) const
SDValue performSraCombine(SDNode *N, DAGCombinerInfo &DCI) const
bool isSelectSupported(SelectSupportKind) const override
bool isZExtFree(Type *Src, Type *Dest) const override
Return true if any actual instruction that defines a value of type FromTy implicitly zero-extends the...
SDValue lowerFEXP2(SDValue Op, SelectionDAG &DAG) const
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
SDValue performSrlCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue lowerFEXP(SDValue Op, SelectionDAG &DAG) const
SDValue getIsLtSmallestNormal(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
SDValue getIsFinite(SelectionDAG &DAG, SDValue Op, SDNodeFlags Flags) const
bool isLoadBitCastBeneficial(EVT, EVT, const SelectionDAG &DAG, const MachineMemOperand &MMO) const final
Return true if the following transform is beneficial: fold (conv (load x)) -> (load (conv*)x) On arch...
std::pair< SDValue, SDValue > splitVector(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HighVT, SelectionDAG &DAG) const
Split a vector value into two parts of types LoVT and HiVT.
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
SDValue LowerFLOGCommon(SDValue Op, SelectionDAG &DAG) const
SDValue foldFreeOpFromSelect(TargetLowering::DAGCombinerInfo &DCI, SDValue N) const
SDValue LowerINT_TO_FP32(SDValue Op, SelectionDAG &DAG, bool Signed) const
bool isFAbsFree(EVT VT) const override
Return true if an fabs operation is free to the point where it is never worthwhile to replace it with...
bool isInt64ImmLegal(SDNode *Val, SelectionDAG &DAG) const
Check whether value Val can be supported by v_mov_b64, for the current target.
SDValue loadStackInputValue(SelectionDAG &DAG, EVT VT, const SDLoc &SL, int64_t Offset) const
Similar to CreateLiveInRegister, except value maybe loaded from a stack slot rather than passed in a ...
SDValue LowerFLOG2(SDValue Op, SelectionDAG &DAG) const
static EVT getEquivalentMemType(LLVMContext &Context, EVT VT)
SDValue LowerCTLS(SDValue Op, SelectionDAG &DAG) const
Split a vector store into multiple scalar stores.
SDValue getSqrtEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps, bool &UseOneConstNR, bool Reciprocal) const override
Hooks for building estimates in place of slower divisions and square roots.
SDValue performTruncateCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerSINT_TO_FP(SDValue Op, SelectionDAG &DAG) const
static SDValue stripBitcast(SDValue Val)
SDValue LowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
SDValue CreateLiveInRegister(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT, const SDLoc &SL, bool RawReg=false) const
Helper function that adds Reg to the LiveIn list of the DAG's MachineFunction.
SDValue SplitVectorStore(SDValue Op, SelectionDAG &DAG) const
Split a vector store into 2 stores of half the vector.
SDValue LowerCTLZ_CTTZ(SDValue Op, SelectionDAG &DAG) const
SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOperations, bool ForCodeSize, NegatibleCost &Cost, unsigned Depth) const override
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > split64BitValue(SDValue Op, SelectionDAG &DAG) const
Return 64-bit value Op as two 32-bit integers.
SDValue performMulCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue getRecipEstimate(SDValue Operand, SelectionDAG &DAG, int Enabled, int &RefinementSteps) const override
Return a reciprocal estimate value for the input operand.
SDValue LowerFNEARBYINT(SDValue Op, SelectionDAG &DAG) const
SDValue LowerSIGN_EXTEND_INREG(SDValue Op, SelectionDAG &DAG) const
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
std::pair< SDValue, SDValue > getScaledLogInput(SelectionDAG &DAG, const SDLoc SL, SDValue Op, SDNodeFlags Flags) const
If denormal handling is required return the scaled input to FLOG2, and the check for denormal range.
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
Selects the correct CCAssignFn for a given CallingConvention value.
bool SimplifyDemandedBitsForTargetNode(SDValue Op, const APInt &OriginalDemandedBits, const APInt &OriginalDemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth) const override
Attempt to simplify any target nodes based on the demanded bits/elts, returning true on success.
static bool allUsesHaveSourceMods(const SDNode *N, unsigned CostThreshold=4)
SDValue LowerFROUNDEVEN(SDValue Op, SelectionDAG &DAG) const
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
Returns true if the target can instruction select the specified FP immediate natively.
static unsigned numBitsUnsigned(SDValue Op, SelectionDAG &DAG)
SDValue lowerFEXPUnsafe(SDValue Op, const SDLoc &SL, SelectionDAG &DAG, SDNodeFlags Flags) const
SDValue LowerFTRUNC(SDValue Op, SelectionDAG &DAG) const
SDValue LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
static bool allowApproxFunc(const SelectionDAG &DAG, SDNodeFlags Flags)
bool ShouldShrinkFPConstant(EVT VT) const override
If true, then instruction selection should seek to shrink the FP constant of the specified type to a ...
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
SDValue performStoreCombine(SDNode *N, DAGCombinerInfo &DCI) const
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue performRcpCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue getLoHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue lowerCTLZResults(SDValue Op, SelectionDAG &DAG) const
SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const
SDValue performFAbsCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT64(SDValue Op, SelectionDAG &DAG, bool Signed) const
static bool shouldFoldFNegIntoSrc(SDNode *FNeg, SDValue FNegSrc)
bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const override
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
SDValue LowerFRINT(SDValue Op, SelectionDAG &DAG) const
SDValue performIntrinsicWOChainCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerUDIVREM(SDValue Op, SelectionDAG &DAG) const
SDValue performMulLoHiCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
void LowerUDIVREM64(SDValue Op, SelectionDAG &DAG, SmallVectorImpl< SDValue > &Results) const
SDValue WidenOrSplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Widen a suitably aligned v3 load.
SDValue LowerDIVREMToFloat(SDValue Op, SelectionDAG &DAG, bool sign) const
std::pair< EVT, EVT > getSplitDestVTs(const EVT &VT, SelectionDAG &DAG) const
Split a vector type into two parts.
SDValue getHiHalf64(SDValue Op, SelectionDAG &DAG) const
SDValue LowerINT_TO_FP16(SDValue Op, SelectionDAG &DAG, EVT FP16Ty) const
SDValue combineFMinMaxLegacyImpl(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, SDValue True, SDValue False, SDValue CC, DAGCombinerInfo &DCI) const
unsigned getVectorIdxWidth(const DataLayout &) const override
Returns the type to be used for the index operand vector operations.
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static const fltSemantics & IEEEdouble()
Definition APFloat.h:305
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
bool bitwiseIsEqual(const APFloat &RHS) const
Definition APFloat.h:1548
opStatus add(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1285
opStatus multiply(const APFloat &RHS, roundingMode RM)
Definition APFloat.h:1303
static APFloat getSmallestNormalized(const fltSemantics &Sem, bool Negative=false)
Returns the smallest (by magnitude) normalized finite number in the given semantics.
Definition APFloat.h:1262
APInt bitcastToAPInt() const
Definition APFloat.h:1475
static APFloat getInf(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Infinity.
Definition APFloat.h:1202
Class for arbitrary precision integers.
Definition APInt.h:78
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1561
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:203
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:255
static APInt getSignedMaxValue(unsigned numBits)
Gets maximum signed value of APInt for a specific bit width.
Definition APInt.h:206
static APInt getSignedMinValue(unsigned numBits)
Gets minimum signed value of APInt for a specific bit width.
Definition APInt.h:216
static APInt getLowBitsSet(unsigned numBits, unsigned loBitsSet)
Constructs an APInt value that has the bottom loBitsSet bits set.
Definition APInt.h:303
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:293
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
const BlockAddress * getBlockAddress() const
CCState - This class holds information needed while lowering arguments and return values.
static CCValAssign getCustomMem(unsigned ValNo, MVT ValVT, int64_t Offset, MVT LocVT, LocInfo HTP)
const APFloat & getValueAPF() const
bool isNegative() const
Return true if the value is negative.
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
Diagnostic information for unsupported feature in backend.
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Definition Function.cpp:360
iterator_range< arg_iterator > args()
Definition Function.h:877
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
This class is used to represent ISD::LOAD nodes.
const SDValue & getBasePtr() const
Machine Value Type.
static auto integer_fixedlen_vector_valuetypes()
uint64_t getScalarSizeInBits() const
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
int getObjectIndexBegin() const
Return the minimum frame object index.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Representation of each machine instruction.
A description of a memory reference used in the backend.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOInvariant
The memory access always returns the same value (or traps).
Flags getFlags() const
Return the raw flags of the source value,.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLVM_ABI bool isLiveIn(Register Reg) const
LLVM_ABI Register getLiveInVirtReg(MCRegister PReg) const
getLiveInVirtReg - If PReg is a live-in physical register, return the corresponding live-in virtual r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
Align getAlign() const
bool isSimple() const
Returns true if the memory operation is neither atomic or volatile.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
const SDValue & getChain() const
bool isInvariant() const
EVT getMemoryVT() const
Return the type of the in-memory value.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
const DebugLoc & getDebugLoc() const
Represents one node in the SelectionDAG.
ArrayRef< SDUse > ops() const
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
SDVTList getVTList() const
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
iterator_range< user_iterator > users()
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
unsigned getOpcode() const
unsigned getNumOperands() const
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
SIModeRegisterDefaults getMode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
LLVM_ABI unsigned ComputeMaxSignificantBits(SDValue Op, unsigned Depth=0) const
Get the upper bound on bit size for this Value Op as a signed integer.
const SDValue & getRoot() const
Return the root tag of the SelectionDAG.
const TargetSubtargetInfo & getSubtarget() const
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
const TargetLowering & getTargetLoweringInfo() const
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
bool isConstantValueOfAnyType(SDValue N) const
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI bool isGuaranteedNotToBeUndefOrPoison(SDValue Op, UndefPoisonKind Kind=UndefPoisonKind::UndefOrPoison, unsigned Depth=0) const
Return true if this function can prove that Op is never poison and, Kind can be used to track poison ...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
const SDValue & setRoot(SDValue N)
Set the current root tag of the SelectionDAG.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
This class is used to represent ISD::STORE nodes.
const SDValue & getBasePtr() const
const SDValue & getValue() const
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
void setMaxDivRemBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum div/rem the backend supports.
bool PredictableSelectIsExpensive
Tells the code generator that select is more expensive than a branch if the branch is usually predict...
virtual bool shouldReduceLoadWidth(SDNode *Load, ISD::LoadExtType ExtTy, EVT NewVT, std::optional< unsigned > ByteOffset=std::nullopt) const
Return true if it is profitable to reduce a load to a smaller type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
unsigned MaxGluedStoresPerMemcpy
Specify max number of store instructions to glue in inlined memcpy.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void addBypassSlowDiv(unsigned int SlowBitWidth, unsigned int FastBitWidth)
Tells the code generator which bitwidths to bypass.
void setMaxLargeFPConvertBitWidthSupported(unsigned SizeInBits)
Set the size in bits of the maximum fp to/from int conversion the backend supports.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
SelectSupportKind
Enum that describes what type of support for selects the target has.
virtual bool allowsMisalignedMemoryAccesses(EVT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *=nullptr) const
Determine if the target supports unaligned memory accesses.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
EVT getShiftAmountTy(EVT LHSTy, const DataLayout &DL) const
Returns the type for the shift amount of a shift opcode.
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
virtual EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const
Return the ValueType of the result of SETCC operations.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
void setSupportsUnalignedAtomics(bool UnalignedSupported)
Sets whether unaligned atomic operations are supported.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
void setMinCmpXchgSizeInBits(unsigned SizeInBits)
Sets the minimum cmpxchg or ll/sc size supported by the backend.
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
unsigned GatherAllAliasesMaxDepth
Depth that GatherAllAliases should continue looking for chain dependencies when trying to find a more...
NegatibleCost
Enum that specifies when a float negation is beneficial.
bool allowsMemoryAccessForAlignment(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
This function returns true if the memory access is aligned or if the target allows this specific unal...
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
void setJumpIsExpensive(bool isExpensive=true)
Tells the code generator not to expand logic operations on comparison predicates into separate sequen...
This class defines information used to lower LLVM code to legal SelectionDAG operators that the targe...
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
bool ShrinkDemandedConstant(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, TargetLoweringOpt &TLO) const
Check to see if the specified operand of the specified instruction is a constant integer.
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
virtual SDValue getNegatedExpression(SDValue Op, SelectionDAG &DAG, bool LegalOps, bool OptForSize, NegatibleCost &Cost, unsigned Depth=0) const
Return the newly negated expression if the cost is not expensive and set the cost in Cost to indicate...
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
TargetLowering(const TargetLowering &)=delete
virtual bool canCreateUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, bool ConsiderFlags, unsigned Depth) const
Return true if Op can create undef or poison from non-undef & non-poison operands.
Primary interface to the complete machine description for the target machine.
TargetSubtargetInfo - Generic base class for all target subtargets.
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:343
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:232
LLVM Value Representation.
Definition Value.h:75
LLVM_ABI StringRef getName() const
Return a constant reference to the value's name.
Definition Value.cpp:319
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
bool isIntrinsicAlwaysUniform(unsigned IntrID)
TargetExtType * isNamedBarrier(const GlobalVariable &GV)
std::optional< APFloat > evaluateRcp(const APFloat &Val)
Evaluate the constant-folded result of v_rcp for Val, accounting for the hardware's denormal flushing...
bool isUniformMMO(const MachineMemOperand *MMO)
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_VS
Used for Mesa vertex shaders, or AMDPAL last shader stage before rasterization (vertex shader if tess...
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_HS
Used for Mesa/AMDPAL hull shaders (= tessellation control shaders).
@ AMDGPU_GS
Used for Mesa/AMDPAL geometry shaders.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ Cold
Attempts to make code in the caller as efficient as possible under the assumption that the call is no...
Definition CallingConv.h:47
@ SPIR_KERNEL
Used for SPIR kernel functions.
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ AMDGPU_ES
Used for AMDPAL shader stage before geometry shader if geometry is in use.
@ AMDGPU_LS
Used for AMDPAL vertex shader if tessellation is in use.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:41
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:829
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:275
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:602
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:789
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADDC
Carry-setting nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:294
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:524
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:863
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:520
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:890
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:586
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:417
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:749
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:280
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:254
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:798
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BRIND
BRIND - Indirect branch.
@ BR_JT
BR_JT - Jumptable branch.
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:543
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:550
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:806
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:247
@ CTLS
Count leading redundant sign bits.
Definition ISDOpcodes.h:802
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:706
@ STRICT_FP16_TO_FP
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:771
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:651
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:616
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ EntryToken
EntryToken - This is the marker used to indicate the start of a region.
Definition ISDOpcodes.h:48
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:224
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:821
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:898
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:729
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:815
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:328
@ INLINEASM_BR
INLINEASM_BR - Branching version of inline asm. Used by asm-goto.
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:936
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:741
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:205
@ ADDE
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:304
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:567
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:53
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:797
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ INLINEASM
INLINEASM - Represents an inline asm block.
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:955
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:62
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:536
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:213
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:558
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
initializer< Ty > init(const Ty &Val)
constexpr double ln2
constexpr double ln10
constexpr float log2ef
Definition MathExtras.h:52
constexpr double log2e
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:577
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1739
InstructionCost Cost
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
LLVM_ABI void ComputeValueVTs(const TargetLowering &TLI, const DataLayout &DL, Type *Ty, SmallVectorImpl< EVT > &ValueVTs, SmallVectorImpl< EVT > *MemVTs=nullptr, SmallVectorImpl< TypeSize > *Offsets=nullptr, TypeSize StartingOffset=TypeSize::getZero())
ComputeValueVTs - Given an LLVM IR type, compute a sequence of EVTs that represent all the individual...
Definition Analysis.cpp:119
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
decltype(auto) get(const PointerIntPair< PointerTy, IntBits, IntType, PtrTraits, Info > &Pair)
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
CombineLevel
Definition DAGCombine.h:15
@ AfterLegalizeDAG
Definition DAGCombine.h:19
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
@ AfterLegalizeTypes
Definition DAGCombine.h:17
To bit_cast(const From &from) noexcept
Definition bit.h:90
@ Mul
Product of integers.
@ Add
Sum of integers.
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
static cl::opt< unsigned > CostThreshold("dfa-cost-threshold", cl::desc("Maximum cost accepted for the transformation"), cl::Hidden, cl::init(50))
APFloat neg(APFloat X)
Returns the negated value of the argument.
Definition APFloat.h:1727
unsigned Log2(Align A)
Returns the log2 of the alignment.
Definition Alignment.h:197
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
MCRegister getRegister() const
unsigned getStackOffset() const
DenormalModeKind Input
Denormal treatment kind for floating point instruction inputs in the default floating-point environme...
@ PreserveSign
The sign of a flushed-to-zero number is preserved in the sign of 0.
static constexpr DenormalMode getPreserveSign()
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
EVT getPow2VectorType(LLVMContext &Context) const
Widens the length of the given vector EVT up to the nearest power of 2 and returns that type.
Definition ValueTypes.h:508
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool bitsGT(EVT VT) const
Return true if this has more bits than VT.
Definition ValueTypes.h:307
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
EVT getHalfSizedIntegerVT(LLVMContext &Context) const
Finds the smallest simple value type that is greater than or equal to half the width of this EVT.
Definition ValueTypes.h:453
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
Definition ValueTypes.h:501
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
EVT getRoundIntegerType(LLVMContext &Context) const
Rounds the bit-width of the given integer EVT up to the nearest power of two (and at least to eight),...
Definition ValueTypes.h:442
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
bool bitsGE(EVT VT) const
Return true if this has no less bits than VT.
Definition ValueTypes.h:315
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isExtended() const
Test if the given EVT is extended (as opposed to being simple).
Definition ValueTypes.h:150
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
LLVM_ABI const fltSemantics & getFltSemantics() const
Returns an APFloat semantics tag appropriate for the value type.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool bitsLE(EVT VT) const
Return true if this has no more bits than VT.
Definition ValueTypes.h:331
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
MVT VT
Legalized type of this argument part.
bool isUnknown() const
Returns true if we don't know any bits.
Definition KnownBits.h:64
KnownBits trunc(unsigned BitWidth) const
Return known bits for a truncation of the value we're tracking.
Definition KnownBits.h:165
KnownBits zext(unsigned BitWidth) const
Return known bits for a zero extension of the value we're tracking.
Definition KnownBits.h:176
unsigned countMaxActiveBits() const
Returns the maximum number of bits needed to represent all possible unsigned values with these known ...
Definition KnownBits.h:310
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
KnownBits sext(unsigned BitWidth) const
Return known bits for a sign extension of the value we're tracking.
Definition KnownBits.h:184
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
Definition KnownBits.h:262
bool isNegative() const
Returns true if this value is known to be negative.
Definition KnownBits.h:103
static LLVM_ABI KnownBits mul(const KnownBits &LHS, const KnownBits &RHS, bool NoUndefSelfMultiply=false)
Compute known bits resulting from multiplying LHS and RHS.
Matching combinators.
This class contains a discriminated union of information about pointers in memory operands,...
LLVM_ABI bool isDereferenceable(unsigned Size, LLVMContext &C, const DataLayout &DL) const
Return true if memory region [V, V+Offset+Size) is known to be dereferenceable.
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
MachinePointerInfo getWithOffset(int64_t O) const
These are IR-level optimization flags that may be propagated to SDNodes.
void setAllowContract(bool b)
bool hasNoSignedZeros() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
LLVM_ABI void AddToWorklist(SDNode *N)
LLVM_ABI SDValue CombineTo(SDNode *N, ArrayRef< SDValue > To, bool AddTo=true)
LLVM_ABI void CommitTargetLoweringOpt(const TargetLoweringOpt &TLO)
A convenience struct that encapsulates a DAG, and two SDValues for returning information from TargetL...