LLVM 24.0.0git
SIISelLowering.cpp
Go to the documentation of this file.
1//===-- SIISelLowering.cpp - SI DAG Lowering Implementation ---------------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// Custom DAG lowering for SI
11//
12//===----------------------------------------------------------------------===//
13
14#include "SIISelLowering.h"
15#include "AMDGPU.h"
16#include "AMDGPUIGroupLP.h"
17#include "AMDGPUInstrInfo.h"
18#include "AMDGPULaneMaskUtils.h"
19#include "AMDGPUMemoryUtils.h"
21#include "AMDGPUTargetMachine.h"
22#include "GCNSubtarget.h"
25#include "SIRegisterInfo.h"
26#include "llvm/ADT/APFloat.h"
27#include "llvm/ADT/APInt.h"
29#include "llvm/ADT/Statistic.h"
44#include "llvm/IR/IRBuilder.h"
46#include "llvm/IR/IntrinsicsAMDGPU.h"
47#include "llvm/IR/IntrinsicsR600.h"
48#include "llvm/IR/MDBuilder.h"
52#include "llvm/Support/ModRef.h"
55#include <optional>
56
57using namespace llvm;
58using namespace llvm::SDPatternMatch;
59
60#define DEBUG_TYPE "si-lower"
61
62STATISTIC(NumTailCalls, "Number of tail calls");
63
64static cl::opt<bool>
65 DisableLoopAlignment("amdgpu-disable-loop-alignment",
66 cl::desc("Do not align and prefetch loops"),
67 cl::init(false));
68
70 "amdgpu-use-divergent-register-indexing", cl::Hidden,
71 cl::desc("Use indirect register addressing for divergent indexes"),
72 cl::init(false));
73
75 return MF.getInfo<SIMachineFunctionInfo>()->getMode().getDenormalFPEnv();
76}
77
82
87
88static unsigned findFirstFreeSGPR(CCState &CCInfo) {
89 unsigned NumSGPRs = AMDGPU::SGPR_32RegClass.getNumRegs();
90 for (unsigned Reg = 0; Reg < NumSGPRs; ++Reg) {
91 if (!CCInfo.isAllocated(AMDGPU::SGPR0 + Reg)) {
92 return AMDGPU::SGPR0 + Reg;
93 }
94 }
95 llvm_unreachable("Cannot allocate sgpr");
96}
97
99 const GCNSubtarget &STI)
100 : AMDGPUTargetLowering(TM, STI, STI), Subtarget(&STI) {
101 addRegisterClass(MVT::i1, &AMDGPU::VReg_1RegClass);
102 addRegisterClass(MVT::i64, &AMDGPU::SReg_64RegClass);
103
104 addRegisterClass(MVT::i32, &AMDGPU::SReg_32RegClass);
105
106 const SIRegisterInfo *TRI = STI.getRegisterInfo();
107 const TargetRegisterClass *V32RegClass =
108 TRI->getDefaultVectorSuperClassForBitWidth(32);
109 addRegisterClass(MVT::f32, V32RegClass);
110
111 addRegisterClass(MVT::v2i32, &AMDGPU::SReg_64RegClass);
112
113 const TargetRegisterClass *V64RegClass =
114 TRI->getDefaultVectorSuperClassForBitWidth(64);
115
116 addRegisterClass(MVT::f64, V64RegClass);
117 addRegisterClass(MVT::v2f32, V64RegClass);
118 addRegisterClass(MVT::Untyped, V64RegClass);
119
120 addRegisterClass(MVT::v3i32, &AMDGPU::SGPR_96RegClass);
121 addRegisterClass(MVT::v3f32, TRI->getDefaultVectorSuperClassForBitWidth(96));
122
123 addRegisterClass(MVT::v2i64, &AMDGPU::SGPR_128RegClass);
124 addRegisterClass(MVT::v2f64, &AMDGPU::SGPR_128RegClass);
125
126 addRegisterClass(MVT::v4i32, &AMDGPU::SGPR_128RegClass);
127 addRegisterClass(MVT::v4f32, TRI->getDefaultVectorSuperClassForBitWidth(128));
128
129 addRegisterClass(MVT::v5i32, &AMDGPU::SGPR_160RegClass);
130 addRegisterClass(MVT::v5f32, TRI->getDefaultVectorSuperClassForBitWidth(160));
131
132 addRegisterClass(MVT::v6i32, &AMDGPU::SGPR_192RegClass);
133 addRegisterClass(MVT::v6f32, TRI->getDefaultVectorSuperClassForBitWidth(192));
134
135 addRegisterClass(MVT::v3i64, &AMDGPU::SGPR_192RegClass);
136 addRegisterClass(MVT::v3f64, TRI->getDefaultVectorSuperClassForBitWidth(192));
137
138 addRegisterClass(MVT::v7i32, &AMDGPU::SGPR_224RegClass);
139 addRegisterClass(MVT::v7f32, TRI->getDefaultVectorSuperClassForBitWidth(224));
140
141 addRegisterClass(MVT::v8i32, &AMDGPU::SGPR_256RegClass);
142 addRegisterClass(MVT::v8f32, TRI->getDefaultVectorSuperClassForBitWidth(256));
143
144 addRegisterClass(MVT::v4i64, &AMDGPU::SGPR_256RegClass);
145 addRegisterClass(MVT::v4f64, TRI->getDefaultVectorSuperClassForBitWidth(256));
146
147 addRegisterClass(MVT::v9i32, &AMDGPU::SGPR_288RegClass);
148 addRegisterClass(MVT::v9f32, TRI->getDefaultVectorSuperClassForBitWidth(288));
149
150 addRegisterClass(MVT::v10i32, &AMDGPU::SGPR_320RegClass);
151 addRegisterClass(MVT::v10f32,
152 TRI->getDefaultVectorSuperClassForBitWidth(320));
153
154 addRegisterClass(MVT::v11i32, &AMDGPU::SGPR_352RegClass);
155 addRegisterClass(MVT::v11f32,
156 TRI->getDefaultVectorSuperClassForBitWidth(352));
157
158 addRegisterClass(MVT::v12i32, &AMDGPU::SGPR_384RegClass);
159 addRegisterClass(MVT::v12f32,
160 TRI->getDefaultVectorSuperClassForBitWidth(384));
161
162 addRegisterClass(MVT::v16i32, &AMDGPU::SGPR_512RegClass);
163 addRegisterClass(MVT::v16f32,
164 TRI->getDefaultVectorSuperClassForBitWidth(512));
165
166 addRegisterClass(MVT::v8i64, &AMDGPU::SGPR_512RegClass);
167 addRegisterClass(MVT::v8f64, TRI->getDefaultVectorSuperClassForBitWidth(512));
168
169 addRegisterClass(MVT::v16i64, &AMDGPU::SGPR_1024RegClass);
170 addRegisterClass(MVT::v16f64,
171 TRI->getDefaultVectorSuperClassForBitWidth(1024));
172
173 if (Subtarget->has16BitInsts()) {
174 if (Subtarget->useRealTrue16Insts()) {
175 addRegisterClass(MVT::i16, &AMDGPU::VGPR_16RegClass);
176 addRegisterClass(MVT::f16, &AMDGPU::VGPR_16RegClass);
177 addRegisterClass(MVT::bf16, &AMDGPU::VGPR_16RegClass);
178 } else {
179 addRegisterClass(MVT::i16, &AMDGPU::SReg_32RegClass);
180 addRegisterClass(MVT::f16, &AMDGPU::SReg_32RegClass);
181 addRegisterClass(MVT::bf16, &AMDGPU::SReg_32RegClass);
182 }
183
184 // Unless there are also VOP3P operations, not operations are really legal.
185 addRegisterClass(MVT::v2i16, &AMDGPU::SReg_32RegClass);
186 addRegisterClass(MVT::v2f16, &AMDGPU::SReg_32RegClass);
187 addRegisterClass(MVT::v2bf16, &AMDGPU::SReg_32RegClass);
188 addRegisterClass(MVT::v4i16, &AMDGPU::SReg_64RegClass);
189 addRegisterClass(MVT::v4f16, &AMDGPU::SReg_64RegClass);
190 addRegisterClass(MVT::v4bf16, &AMDGPU::SReg_64RegClass);
191 addRegisterClass(MVT::v8i16, &AMDGPU::SGPR_128RegClass);
192 addRegisterClass(MVT::v8f16, &AMDGPU::SGPR_128RegClass);
193 addRegisterClass(MVT::v8bf16, &AMDGPU::SGPR_128RegClass);
194 addRegisterClass(MVT::v16i16, &AMDGPU::SGPR_256RegClass);
195 addRegisterClass(MVT::v16f16, &AMDGPU::SGPR_256RegClass);
196 addRegisterClass(MVT::v16bf16, &AMDGPU::SGPR_256RegClass);
197 addRegisterClass(MVT::v32i16, &AMDGPU::SGPR_512RegClass);
198 addRegisterClass(MVT::v32f16, &AMDGPU::SGPR_512RegClass);
199 addRegisterClass(MVT::v32bf16, &AMDGPU::SGPR_512RegClass);
200 }
201
202 addRegisterClass(MVT::v32i32, &AMDGPU::VReg_1024RegClass);
203 addRegisterClass(MVT::v32f32,
204 TRI->getDefaultVectorSuperClassForBitWidth(1024));
205
206 computeRegisterProperties(Subtarget->getRegisterInfo());
207
210
211 // The boolean content concept here is too inflexible. Compares only ever
212 // really produce a 1-bit result. Any copy/extend from these will turn into a
213 // select, and zext/1 or sext/-1 are equally cheap. Arbitrarily choose 0/1, as
214 // it's what most targets use.
217
218 // We need to custom lower vector stores from local memory
220 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
221 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
222 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32,
223 MVT::i1, MVT::v32i32},
224 Custom);
225
227 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
228 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
229 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32,
230 MVT::i1, MVT::v32i32},
231 Custom);
232
233 if (isTypeLegal(MVT::bf16)) {
234 for (unsigned Opc :
243 ISD::SETCC}) {
244 setOperationAction(Opc, MVT::bf16, Promote);
245 }
246
247 // Only targets with packed bf16 instructions, e.g. gfx13.
248 if (Subtarget->hasBF16PackedInsts()) {
249 // Don't use Expand for fsub - the DAG combiner will undo fadd+fneg back
250 // to fsub, causing a libcall (which doesn't exist for bf16). Instead,
251 // directly expand to widened v2bf16 operations.
253 // Promote scalar operations to a v2bf16 operation with an unused high
254 // lane.
255 for (unsigned Opc : {ISD::FADD, ISD::FMUL, ISD::FMA, ISD::FMAXNUM,
257 AddPromotedToType(Opc, MVT::bf16, MVT::v2bf16);
258 }
259
261
263 AddPromotedToType(ISD::SELECT, MVT::bf16, MVT::i16);
264
268
269 // We only need to custom lower because we can't specify an action for bf16
270 // sources.
273 }
274
275 setTruncStoreAction(MVT::v2i32, MVT::v2i16, Expand);
276 setTruncStoreAction(MVT::v3i32, MVT::v3i16, Expand);
277 setTruncStoreAction(MVT::v4i32, MVT::v4i16, Expand);
278 setTruncStoreAction(MVT::v8i32, MVT::v8i16, Expand);
279 setTruncStoreAction(MVT::v16i32, MVT::v16i16, Expand);
280 setTruncStoreAction(MVT::v32i32, MVT::v32i16, Expand);
281 setTruncStoreAction(MVT::v2i32, MVT::v2i8, Expand);
282 setTruncStoreAction(MVT::v4i32, MVT::v4i8, Expand);
283 setTruncStoreAction(MVT::v8i32, MVT::v8i8, Expand);
284 setTruncStoreAction(MVT::v16i32, MVT::v16i8, Expand);
285 setTruncStoreAction(MVT::v32i32, MVT::v32i8, Expand);
286 setTruncStoreAction(MVT::v2i16, MVT::v2i8, Expand);
287 setTruncStoreAction(MVT::v4i16, MVT::v4i8, Expand);
288 setTruncStoreAction(MVT::v8i16, MVT::v8i8, Expand);
289 setTruncStoreAction(MVT::v16i16, MVT::v16i8, Expand);
290 setTruncStoreAction(MVT::v32i16, MVT::v32i8, Expand);
291
292 setTruncStoreAction(MVT::v3i64, MVT::v3i16, Expand);
293 setTruncStoreAction(MVT::v3i64, MVT::v3i32, Expand);
294 setTruncStoreAction(MVT::v4i64, MVT::v4i8, Expand);
295 setTruncStoreAction(MVT::v8i64, MVT::v8i8, Expand);
296 setTruncStoreAction(MVT::v8i64, MVT::v8i16, Expand);
297 setTruncStoreAction(MVT::v8i64, MVT::v8i32, Expand);
298 setTruncStoreAction(MVT::v16i64, MVT::v16i32, Expand);
299
300 setOperationAction(ISD::GlobalAddress, {MVT::i32, MVT::i64}, Custom);
301 setOperationAction(ISD::BlockAddress, {MVT::i32, MVT::i64}, Custom);
302 setOperationAction(ISD::ExternalSymbol, {MVT::i32, MVT::i64}, Custom);
303
307 AddPromotedToType(ISD::SELECT, MVT::f64, MVT::i64);
308
309 setOperationAction(ISD::FSQRT, {MVT::f32, MVT::f64}, Custom);
310
312 {MVT::f32, MVT::i32, MVT::i64, MVT::f64, MVT::i1}, Expand);
313
315 setOperationAction(ISD::SETCC, {MVT::v2i1, MVT::v4i1}, Expand);
316 AddPromotedToType(ISD::SETCC, MVT::i1, MVT::i32);
317
319 {MVT::v2i32, MVT::v3i32, MVT::v4i32, MVT::v5i32,
320 MVT::v6i32, MVT::v7i32, MVT::v8i32, MVT::v9i32,
321 MVT::v10i32, MVT::v11i32, MVT::v12i32, MVT::v16i32},
322 Expand);
324 {MVT::v2f32, MVT::v3f32, MVT::v4f32, MVT::v5f32,
325 MVT::v6f32, MVT::v7f32, MVT::v8f32, MVT::v9f32,
326 MVT::v10f32, MVT::v11f32, MVT::v12f32, MVT::v16f32},
327 Expand);
328
330 {MVT::v2i1, MVT::v4i1, MVT::v2i8, MVT::v4i8, MVT::v2i16,
331 MVT::v3i16, MVT::v4i16, MVT::Other},
332 Custom);
333
336 {MVT::i1, MVT::i32, MVT::i64, MVT::f32, MVT::f64}, Expand);
337
340
343
345 Expand);
346
348
349 // We only support LOAD/STORE and vector manipulation ops for vectors
350 // with > 4 elements.
351 for (MVT VT :
352 {MVT::v8i32, MVT::v8f32, MVT::v9i32, MVT::v9f32, MVT::v10i32,
353 MVT::v10f32, MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32,
354 MVT::v16i32, MVT::v16f32, MVT::v2i64, MVT::v2f64, MVT::v4i16,
355 MVT::v4f16, MVT::v4bf16, MVT::v3i64, MVT::v3f64, MVT::v6i32,
356 MVT::v6f32, MVT::v4i64, MVT::v4f64, MVT::v8i64, MVT::v8f64,
357 MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16, MVT::v16f16,
358 MVT::v16bf16, MVT::v16i64, MVT::v16f64, MVT::v32i32, MVT::v32f32,
359 MVT::v32i16, MVT::v32f16, MVT::v32bf16}) {
360 for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op) {
361 switch (Op) {
362 case ISD::LOAD:
363 case ISD::STORE:
364 case ISD::ATOMIC_LOAD:
367 case ISD::BITCAST:
368 case ISD::UNDEF:
369 case ISD::POISON:
373 case ISD::IS_FPCLASS:
374 break;
379 break;
380 default:
382 break;
383 }
384 }
385 }
386
388
389 // TODO: For dynamic 64-bit vector inserts/extracts, should emit a pseudo that
390 // is expanded to avoid having two separate loops in case the index is a VGPR.
391
392 // Most operations are naturally 32-bit vector operations. We only support
393 // load and store of i64 vectors, so promote v2i64 vector operations to v4i32.
394 for (MVT Vec64 : {MVT::v2i64, MVT::v2f64}) {
396 AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v4i32);
397
399 AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v4i32);
400
402 AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v4i32);
403
405 AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v4i32);
406 }
407
408 for (MVT Vec64 : {MVT::v3i64, MVT::v3f64}) {
410 AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v6i32);
411
413 AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v6i32);
414
416 AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v6i32);
417
419 AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v6i32);
420 }
421
422 for (MVT Vec64 : {MVT::v4i64, MVT::v4f64}) {
424 AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v8i32);
425
427 AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v8i32);
428
430 AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v8i32);
431
433 AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v8i32);
434 }
435
436 for (MVT Vec64 : {MVT::v8i64, MVT::v8f64}) {
438 AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v16i32);
439
441 AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v16i32);
442
444 AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v16i32);
445
447 AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v16i32);
448 }
449
450 for (MVT Vec64 : {MVT::v16i64, MVT::v16f64}) {
452 AddPromotedToType(ISD::BUILD_VECTOR, Vec64, MVT::v32i32);
453
455 AddPromotedToType(ISD::EXTRACT_VECTOR_ELT, Vec64, MVT::v32i32);
456
458 AddPromotedToType(ISD::INSERT_VECTOR_ELT, Vec64, MVT::v32i32);
459
461 AddPromotedToType(ISD::SCALAR_TO_VECTOR, Vec64, MVT::v32i32);
462 }
463
465 {MVT::v4i32, MVT::v4f32, MVT::v8i32, MVT::v8f32,
466 MVT::v16i32, MVT::v16f32, MVT::v32i32, MVT::v32f32},
467 Custom);
468
469 if (Subtarget->hasPkMovB32()) {
470 // TODO: 16-bit element vectors should be legal with even aligned elements.
471 // TODO: Can be legal with wider source types than the result with
472 // subregister extracts.
473 setOperationAction(ISD::VECTOR_SHUFFLE, {MVT::v2i32, MVT::v2f32}, Legal);
474 }
475
477 // Prevent SELECT v2i32 from being implemented with the above bitwise ops and
478 // instead lower to cndmask in SITargetLowering::LowerSELECT().
480 // Enable MatchRotate to produce ISD::ROTR, which is later transformed to
481 // alignbit.
482 setOperationAction(ISD::ROTR, MVT::v2i32, Custom);
483
484 setOperationAction(ISD::BUILD_VECTOR, {MVT::v4f16, MVT::v4i16, MVT::v4bf16},
485 Custom);
486
487 // Avoid stack access for these.
488 // TODO: Generalize to more vector types.
490 {MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::v2i8, MVT::v4i8,
491 MVT::v8i8, MVT::v4i16, MVT::v4f16, MVT::v4bf16},
492 Custom);
493
494 // Deal with vec3 vector operations when widened to vec4.
496 {MVT::v3i32, MVT::v3f32, MVT::v4i32, MVT::v4f32}, Custom);
497
498 // Deal with vec5/6/7 vector operations when widened to vec8.
500 {MVT::v5i32, MVT::v5f32, MVT::v6i32, MVT::v6f32,
501 MVT::v7i32, MVT::v7f32, MVT::v8i32, MVT::v8f32,
502 MVT::v9i32, MVT::v9f32, MVT::v10i32, MVT::v10f32,
503 MVT::v11i32, MVT::v11f32, MVT::v12i32, MVT::v12f32},
504 Custom);
505
506 // BUFFER/FLAT_ATOMIC_CMP_SWAP on GCN GPUs needs input marshalling,
507 // and output demarshalling
508 setOperationAction(ISD::ATOMIC_CMP_SWAP, {MVT::i32, MVT::i64}, Custom);
509
510 // We can't return success/failure, only the old value,
511 // let LLVM add the comparison
513 Expand);
514
515 setOperationAction(ISD::ADDRSPACECAST, {MVT::i32, MVT::i64}, Custom);
516
517 setOperationAction(ISD::BITREVERSE, {MVT::i32, MVT::i64}, Legal);
518
519 // FIXME: This should be narrowed to i32, but that only happens if i64 is
520 // illegal.
521 // FIXME: Should lower sub-i32 bswaps to bit-ops without v_perm_b32.
522 setOperationAction(ISD::BSWAP, {MVT::i64, MVT::i32}, Legal);
523
524 // On SI this is s_memtime and s_memrealtime on VI.
526
527 if (Subtarget->hasSMemRealTime() ||
528 Subtarget->getGeneration() >= AMDGPUSubtarget::GFX11)
531
532 if (Subtarget->has16BitInsts()) {
535 setOperationAction(ISD::IS_FPCLASS, {MVT::f16, MVT::f32, MVT::f64}, Legal);
538 } else {
540 }
541
542 if (Subtarget->hasMadMacF32Insts())
544
548
549 // We only really have 32-bit BFE instructions (and 16-bit on VI).
550 //
551 // On SI+ there are 64-bit BFEs, but they are scalar only and there isn't any
552 // effort to match them now. We want this to be false for i64 cases when the
553 // extraction isn't restricted to the upper or lower half. Ideally we would
554 // have some pass reduce 64-bit extracts to 32-bit if possible. Extracts that
555 // span the midpoint are probably relatively rare, so don't worry about them
556 // for now.
558
559 // Clamp modifier on add/sub
560 if (Subtarget->hasIntClamp())
562
563 if (Subtarget->hasAddNoCarryInsts())
564 setOperationAction({ISD::SADDSAT, ISD::SSUBSAT}, {MVT::i16, MVT::i32},
565 Legal);
566
567 // Do not have s_{min|max}_*f64 instruction f64 will only be lowered to
568 // v_{min|max}_*f64
569 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
572 {MVT::f64, MVT::f32}, Legal);
573 } else {
576 {MVT::f64, MVT::f32}, Custom);
577 // These are really only legal for ieee_mode functions. We should be
578 // avoiding them for functions that don't have ieee_mode enabled, so just
579 // say they are legal.
581 {MVT::f64, MVT::f32}, Legal);
582 }
583
584 if (Subtarget->haveRoundOpsF64())
586 Legal);
587 else
589 MVT::f64, Custom);
590
592 setOperationAction({ISD::FLDEXP, ISD::STRICT_FLDEXP}, {MVT::f32, MVT::f64},
593 Legal);
594 setOperationAction(ISD::FFREXP, {MVT::f32, MVT::f64}, Custom);
595
598
599 setOperationAction(ISD::BF16_TO_FP, {MVT::i16, MVT::f32, MVT::f64}, Expand);
600 setOperationAction(ISD::FP_TO_BF16, {MVT::i16, MVT::f32, MVT::f64}, Expand);
601
603 Custom);
605 Custom);
607 Custom);
608
609 // Custom lower these because we can't specify a rule based on an illegal
610 // source bf16.
613
614 if (Subtarget->has16BitInsts()) {
617 MVT::i16, Legal);
618
619 AddPromotedToType(ISD::SIGN_EXTEND, MVT::i16, MVT::i32);
620
622 MVT::i16, Expand);
623
627 ISD::CTPOP},
628 MVT::i16, Promote);
629
631
632 setTruncStoreAction(MVT::i64, MVT::i16, Expand);
633
635 AddPromotedToType(ISD::FP16_TO_FP, MVT::i16, MVT::i32);
637 AddPromotedToType(ISD::FP_TO_FP16, MVT::i16, MVT::i32);
638
643
645
646 // F16 - Constant Actions.
649
650 // F16 - Load/Store Actions.
652 AddPromotedToType(ISD::LOAD, MVT::f16, MVT::i16);
654 AddPromotedToType(ISD::STORE, MVT::f16, MVT::i16);
655
656 // BF16 - Load/Store Actions.
658 AddPromotedToType(ISD::LOAD, MVT::bf16, MVT::i16);
660 AddPromotedToType(ISD::STORE, MVT::bf16, MVT::i16);
661
662 // F16 - VOP1 Actions.
665 MVT::f16, Custom);
666
667 // BF16 - VOP1 Actions.
668 if (Subtarget->hasBF16TransInsts())
670
671 // F16 - VOP2 Actions.
672 setOperationAction({ISD::BR_CC, ISD::SELECT_CC}, {MVT::f16, MVT::bf16},
673 Expand);
677
678 // F16 - VOP3 Actions.
680 if (STI.hasMadF16())
682
683 for (MVT VT :
684 {MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::v4i16, MVT::v4f16,
685 MVT::v4bf16, MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16,
686 MVT::v16f16, MVT::v16bf16, MVT::v32i16, MVT::v32f16}) {
687 for (unsigned Op = 0; Op < ISD::BUILTIN_OP_END; ++Op) {
688 switch (Op) {
689 case ISD::LOAD:
690 case ISD::STORE:
691 case ISD::ATOMIC_LOAD:
694 case ISD::BITCAST:
695 case ISD::UNDEF:
696 case ISD::POISON:
701 case ISD::IS_FPCLASS:
702 break;
705 case ISD::FSIN:
706 case ISD::FCOS:
708 break;
709 default:
711 break;
712 }
713 }
714 }
715
716 // v_perm_b32 can handle either of these.
717 setOperationAction(ISD::BSWAP, {MVT::i16, MVT::v2i16}, Legal);
719
720 // Legalize vector types for sat conversions to select v_cvt_pk_[iu]16_f32.
721 if (Subtarget->hasVCvtPkIU16F32())
724 {MVT::v2i16, MVT::v4i16, MVT::v8i16, MVT::v16i16, MVT::v32i16},
725 Custom);
726
727 // XXX - Do these do anything? Vector constants turn into build_vector.
728 setOperationAction(ISD::Constant, {MVT::v2i16, MVT::v2f16}, Legal);
729
731 {MVT::v2i16, MVT::v2f16, MVT::v2bf16}, Legal);
732
734 AddPromotedToType(ISD::STORE, MVT::v2i16, MVT::i32);
736 AddPromotedToType(ISD::STORE, MVT::v2f16, MVT::i32);
737
739 AddPromotedToType(ISD::LOAD, MVT::v2i16, MVT::i32);
741 AddPromotedToType(ISD::LOAD, MVT::v2f16, MVT::i32);
742
743 setOperationAction(ISD::AND, MVT::v2i16, Promote);
744 AddPromotedToType(ISD::AND, MVT::v2i16, MVT::i32);
745 setOperationAction(ISD::OR, MVT::v2i16, Promote);
746 AddPromotedToType(ISD::OR, MVT::v2i16, MVT::i32);
747 setOperationAction(ISD::XOR, MVT::v2i16, Promote);
748 AddPromotedToType(ISD::XOR, MVT::v2i16, MVT::i32);
749
751 AddPromotedToType(ISD::LOAD, MVT::v4i16, MVT::v2i32);
753 AddPromotedToType(ISD::LOAD, MVT::v4f16, MVT::v2i32);
754 setOperationAction(ISD::LOAD, MVT::v4bf16, Promote);
755 AddPromotedToType(ISD::LOAD, MVT::v4bf16, MVT::v2i32);
756
758 AddPromotedToType(ISD::STORE, MVT::v4i16, MVT::v2i32);
760 AddPromotedToType(ISD::STORE, MVT::v4f16, MVT::v2i32);
762 AddPromotedToType(ISD::STORE, MVT::v4bf16, MVT::v2i32);
763
765 AddPromotedToType(ISD::LOAD, MVT::v8i16, MVT::v4i32);
767 AddPromotedToType(ISD::LOAD, MVT::v8f16, MVT::v4i32);
768 setOperationAction(ISD::LOAD, MVT::v8bf16, Promote);
769 AddPromotedToType(ISD::LOAD, MVT::v8bf16, MVT::v4i32);
770
772 AddPromotedToType(ISD::STORE, MVT::v4i16, MVT::v2i32);
774 AddPromotedToType(ISD::STORE, MVT::v4f16, MVT::v2i32);
775
777 AddPromotedToType(ISD::STORE, MVT::v8i16, MVT::v4i32);
779 AddPromotedToType(ISD::STORE, MVT::v8f16, MVT::v4i32);
781 AddPromotedToType(ISD::STORE, MVT::v8bf16, MVT::v4i32);
782
783 setOperationAction(ISD::LOAD, MVT::v16i16, Promote);
784 AddPromotedToType(ISD::LOAD, MVT::v16i16, MVT::v8i32);
785 setOperationAction(ISD::LOAD, MVT::v16f16, Promote);
786 AddPromotedToType(ISD::LOAD, MVT::v16f16, MVT::v8i32);
787 setOperationAction(ISD::LOAD, MVT::v16bf16, Promote);
788 AddPromotedToType(ISD::LOAD, MVT::v16bf16, MVT::v8i32);
789
791 AddPromotedToType(ISD::STORE, MVT::v16i16, MVT::v8i32);
793 AddPromotedToType(ISD::STORE, MVT::v16f16, MVT::v8i32);
794 setOperationAction(ISD::STORE, MVT::v16bf16, Promote);
795 AddPromotedToType(ISD::STORE, MVT::v16bf16, MVT::v8i32);
796
797 setOperationAction(ISD::LOAD, MVT::v32i16, Promote);
798 AddPromotedToType(ISD::LOAD, MVT::v32i16, MVT::v16i32);
799 setOperationAction(ISD::LOAD, MVT::v32f16, Promote);
800 AddPromotedToType(ISD::LOAD, MVT::v32f16, MVT::v16i32);
801 setOperationAction(ISD::LOAD, MVT::v32bf16, Promote);
802 AddPromotedToType(ISD::LOAD, MVT::v32bf16, MVT::v16i32);
803
805 AddPromotedToType(ISD::STORE, MVT::v32i16, MVT::v16i32);
807 AddPromotedToType(ISD::STORE, MVT::v32f16, MVT::v16i32);
808 setOperationAction(ISD::STORE, MVT::v32bf16, Promote);
809 AddPromotedToType(ISD::STORE, MVT::v32bf16, MVT::v16i32);
810
812 MVT::v2i32, Expand);
814
816 MVT::v4i32, Expand);
817
819 MVT::v8i32, Expand);
820
821 setOperationAction(ISD::BUILD_VECTOR, {MVT::v2i16, MVT::v2f16, MVT::v2bf16},
822 Subtarget->hasVOP3PInsts() ? Legal : Custom);
823
824 setOperationAction(ISD::FNEG, {MVT::v2f16, MVT::v2bf16}, Legal);
825 // This isn't really legal, but this avoids the legalizer unrolling it (and
826 // allows matching fneg (fabs x) patterns)
827 setOperationAction(ISD::FABS, {MVT::v2f16, MVT::v2bf16}, Legal);
828
829 // Can do this in one BFI plus a constant materialize.
831 {MVT::v2f16, MVT::v2bf16, MVT::v4f16, MVT::v4bf16,
832 MVT::v8f16, MVT::v8bf16, MVT::v16f16, MVT::v16bf16,
833 MVT::v32f16, MVT::v32bf16},
834 Custom);
835 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
838 MVT::f16, Legal);
839
842 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16}, Custom);
843 } else {
846 MVT::f16, Custom);
847
849 Legal);
850
853 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
854 Custom);
855
857 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
858 Expand);
859 }
860
861 for (MVT Vec16 :
862 {MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::v16i16, MVT::v16f16,
863 MVT::v16bf16, MVT::v32i16, MVT::v32f16, MVT::v32bf16}) {
866 Vec16, Custom);
868 }
869 }
870
871 if (Subtarget->hasVOP3PInsts()) {
875 MVT::v2i16, Legal);
876
879 MVT::v2f16, Legal);
880
882 {MVT::v2i16, MVT::v2f16, MVT::v2bf16}, Custom);
883
885 {MVT::v4f16, MVT::v4i16, MVT::v4bf16, MVT::v8f16,
886 MVT::v8i16, MVT::v8bf16, MVT::v16f16, MVT::v16i16,
887 MVT::v16bf16, MVT::v32f16, MVT::v32i16, MVT::v32bf16},
888 Custom);
889
890 for (MVT VT : {MVT::v4i16, MVT::v8i16, MVT::v16i16, MVT::v32i16})
891 // Split vector operations.
896 VT, Custom);
897
898 for (MVT VT : {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16})
899 // Split vector operations.
902 VT, Custom);
903
904 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
907 MVT::v2f16, Legal);
908 } else {
910 Legal);
911
914 {MVT::v2f16, MVT::v4f16}, Custom);
915 }
916 setOperationAction(ISD::FEXP, MVT::v2f16, Custom);
917 setOperationAction(ISD::SELECT, {MVT::v4i16, MVT::v4f16, MVT::v4bf16},
918 Custom);
919
920 if (Subtarget->hasBF16PackedInsts()) {
924 MVT::v2bf16, Legal);
925
926 for (MVT VT : {MVT::v4bf16, MVT::v8bf16, MVT::v16bf16, MVT::v32bf16})
927 // Split vector operations.
931 VT, Custom);
932 }
933
934 if (Subtarget->hasAnyPackedFP32Ops()) {
936 MVT::v2f32, Legal);
938 {MVT::v4f32, MVT::v8f32, MVT::v16f32, MVT::v32f32},
939 Custom);
940 }
941 if (Subtarget->hasAnyPackedFP64Ops()) {
944 MVT::v2f64, Legal);
947 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64}, Custom);
948
949 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
952 MVT::v2f64, Legal);
953
956 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64}, Custom);
957 } else {
959 Legal);
962 MVT::v2f64, Custom);
965 {MVT::v4f64, MVT::v8f64, MVT::v16f64, MVT::v32f64},
966 Custom);
967 }
968 }
969
970 if (Subtarget->hasAnyPackedU64Ops()) {
972 MVT::v2i64, Legal);
974 {MVT::v4i64, MVT::v8i64, MVT::v16i64, MVT::v32i64},
975 Custom);
976 }
977 }
978
980
981 if (Subtarget->has16BitInsts()) {
983 AddPromotedToType(ISD::SELECT, MVT::v2i16, MVT::i32);
985 AddPromotedToType(ISD::SELECT, MVT::v2f16, MVT::i32);
987 AddPromotedToType(ISD::SELECT, MVT::v2bf16, MVT::i32);
988 } else {
989 // Legalization hack.
990 setOperationAction(ISD::SELECT, {MVT::v2i16, MVT::v2f16}, Custom);
991
993 }
994
996 {MVT::v4i16, MVT::v4f16, MVT::v4bf16, MVT::v2i8, MVT::v4i8,
997 MVT::v8i8, MVT::v8i16, MVT::v8f16, MVT::v8bf16,
998 MVT::v16i16, MVT::v16f16, MVT::v16bf16, MVT::v32i16,
999 MVT::v32f16, MVT::v32bf16},
1000 Custom);
1001
1003
1004 if (Subtarget->useVMulU64Inst())
1005 setOperationAction(ISD::MUL, MVT::i64, Legal);
1006 else if (Subtarget->hasScalarSMulU64())
1008
1009 if (Subtarget->hasMad64_32())
1011
1012 if (Subtarget->hasSafeSmemPrefetch() || Subtarget->hasVmemPrefInsts())
1014
1015 if (Subtarget->hasIEEEMinimumMaximumInsts()) {
1017 {MVT::f16, MVT::f32, MVT::f64, MVT::v2f16}, Legal);
1018 } else {
1019 // FIXME: For nnan fmaximum, emit the fmaximum3 instead of fmaxnum
1020 if (Subtarget->hasMinimum3Maximum3F32())
1022
1023 if (Subtarget->hasMinimum3Maximum3PKF16()) {
1025
1026 // If only the vector form is available, we need to widen to a vector.
1027 if (!Subtarget->hasMinimum3Maximum3F16())
1029 MVT::v2f16);
1030 }
1031 }
1032
1033 if (Subtarget->hasVOP3PInsts()) {
1034 // We want to break these into v2f16 pieces, not scalarize.
1036 {MVT::v4f16, MVT::v8f16, MVT::v16f16, MVT::v32f16},
1037 Custom);
1038 }
1039
1040 if (Subtarget->useMinMaxI64Insts())
1042 Legal);
1043
1045 {MVT::Other, MVT::f32, MVT::v4f32, MVT::i16, MVT::f16,
1046 MVT::bf16, MVT::v2i16, MVT::v2f16, MVT::v2bf16, MVT::i128,
1047 MVT::i8},
1048 Custom);
1049
1051 {MVT::v2f16, MVT::v2i16, MVT::v2bf16, MVT::v3f16,
1052 MVT::v3i16, MVT::v4f16, MVT::v4i16, MVT::v4bf16,
1053 MVT::v8i16, MVT::v8f16, MVT::v8bf16, MVT::Other, MVT::f16,
1054 MVT::i16, MVT::bf16, MVT::i8, MVT::i128},
1055 Custom);
1056
1057 // The s_buffer_load intrinsics accept any result type in IR, but only a few
1058 // of them can be selected. Mark the remaining illegal result types Custom so
1059 // ReplaceNodeResults gets a chance to diagnose them instead of letting the
1060 // type legalizer abort. Its INTRINSIC_WO_CHAIN case dispatches on the
1061 // intrinsic ID, but INTRINSIC_W_CHAIN does not, so remember the types added
1062 // here to keep other chained intrinsics on generic legalization.
1063 for (MVT VT : MVT::all_valuetypes()) {
1064 if (VT.isScalableVector() || isTypeLegal(VT))
1065 continue;
1069 SBufferLoadDiagnosticVTs.set(VT.SimpleTy);
1070 }
1071 }
1072
1074 {MVT::Other, MVT::v2i16, MVT::v2f16, MVT::v2bf16,
1075 MVT::v3i16, MVT::v3f16, MVT::v4f16, MVT::v4i16,
1076 MVT::v4bf16, MVT::v8i16, MVT::v8f16, MVT::v8bf16,
1077 MVT::f16, MVT::i16, MVT::bf16, MVT::i8, MVT::i128},
1078 Custom);
1079
1085
1086 // TODO: Could move this to custom lowering, could benefit from combines on
1087 // extract of relevant bits.
1089
1091
1092 if (Subtarget->hasBF16ConversionInsts()) {
1094 {MVT::bf16, MVT::v2bf16}, Custom);
1096 }
1097
1098 if (Subtarget->hasBF16TransInsts()) {
1100 }
1101
1102 const bool HasE5M3ConversionInsts =
1103 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
1104 if (Subtarget->hasOCPFP8ConversionInsts() || HasE5M3ConversionInsts) {
1105 setOperationAction(ISD::CONVERT_FROM_ARBITRARY_FP, {MVT::f32, MVT::v2f32},
1106 Custom);
1108
1109 // i8 result promotes to i16, wider vectors split down to v2i8, and v2i8 is
1110 // handled in ReplaceNodeResults before the legalizer splits it per lane.
1111 setOperationAction(ISD::CONVERT_TO_ARBITRARY_FP, {MVT::i16, MVT::v2i8},
1112 Custom);
1113 }
1114
1115 if (Subtarget->hasFP8F16ConversionInsts()) {
1116 setOperationAction(ISD::CONVERT_FROM_ARBITRARY_FP, {MVT::f16, MVT::v2f16},
1117 Custom);
1118 }
1119
1120 if (Subtarget->hasCvtPkF16F32Inst()) {
1122 {MVT::v2f16, MVT::v4f16, MVT::v8f16, MVT::v16f16},
1123 Custom);
1124 }
1125
1128 ISD::SUB,
1129 ISD::MUL,
1130 ISD::FADD,
1131 ISD::FSUB,
1132 ISD::FDIV,
1133 ISD::FMUL,
1142 ISD::FMA,
1143 ISD::ABS,
1144 ISD::SMIN,
1145 ISD::SMAX,
1146 ISD::UMIN,
1147 ISD::UMAX,
1148 ISD::SETCC,
1150 ISD::SMIN,
1151 ISD::SMAX,
1152 ISD::UMIN,
1153 ISD::UMAX,
1156 ISD::AND,
1157 ISD::OR,
1158 ISD::XOR,
1159 ISD::SHL,
1160 ISD::SRL,
1161 ISD::SRA,
1162 ISD::FSHR,
1173
1174 if (Subtarget->has16BitInsts() && !Subtarget->hasMed3_16())
1176
1177 // All memory operations. Some folding on the pointer operand is done to help
1178 // matching the constant offsets in the addressing modes.
1180 ISD::STORE,
1205
1206 // FIXME: In other contexts we pretend this is a per-function property.
1208
1210}
1211
1212const GCNSubtarget *SITargetLowering::getSubtarget() const { return Subtarget; }
1213
1215 static const MCPhysReg RCRegs[] = {AMDGPU::MODE};
1216 return RCRegs;
1217}
1218
1219//===----------------------------------------------------------------------===//
1220// TargetLowering queries
1221//===----------------------------------------------------------------------===//
1222
1223// v_mad_mix* support a conversion from f16 to f32.
1224//
1225// There is only one special case when denormals are enabled we don't currently,
1226// where this is OK to use.
1227bool SITargetLowering::isFPExtFoldable(const SelectionDAG &DAG, unsigned Opcode,
1228 EVT DestVT, EVT SrcVT) const {
1229 return DestVT.getScalarType() == MVT::f32 &&
1230 ((((Opcode == ISD::FMAD && Subtarget->hasMadMixInsts()) ||
1231 (Opcode == ISD::FMA && Subtarget->hasFmaMixInsts())) &&
1232 SrcVT.getScalarType() == MVT::f16) ||
1233 (Opcode == ISD::FMA && Subtarget->hasFmaMixBF16Insts() &&
1234 SrcVT.getScalarType() == MVT::bf16)) &&
1235 // TODO: This probably only requires no input flushing?
1237}
1238
1240 LLT DestTy, LLT SrcTy) const {
1241 return ((Opcode == TargetOpcode::G_FMAD && Subtarget->hasMadMixInsts()) ||
1242 (Opcode == TargetOpcode::G_FMA && Subtarget->hasFmaMixInsts())) &&
1243 DestTy.getScalarSizeInBits() == 32 &&
1244 SrcTy.getScalarSizeInBits() == 16 &&
1245 // TODO: This probably only requires no input flushing?
1246 denormalModeIsFlushAllF32(*MI.getMF());
1247}
1248
1250 // SI has some legal vector types, but no legal vector operations. Say no
1251 // shuffles are legal in order to prefer scalarizing some vector operations.
1252 return false;
1253}
1254
1256 CallingConv::ID CC,
1257 EVT VT) const {
1259 return TargetLowering::getRegisterTypeForCallingConv(Context, CC, VT);
1260
1261 if (VT.isVector()) {
1262 EVT ScalarVT = VT.getScalarType();
1263 unsigned Size = ScalarVT.getSizeInBits();
1264 if (Size == 16) {
1265 return Subtarget->has16BitInsts()
1266 ? MVT::getVectorVT(ScalarVT.getSimpleVT(), 2)
1267 : MVT::i32;
1268 }
1269
1270 if (Size < 16)
1271 return Subtarget->has16BitInsts() ? MVT::i16 : MVT::i32;
1272 return Size == 32 ? ScalarVT.getSimpleVT() : MVT::i32;
1273 }
1274
1275 if (!Subtarget->has16BitInsts() && VT.getSizeInBits() == 16)
1276 return MVT::i32;
1277
1278 if (VT.getSizeInBits() > 32)
1279 return MVT::i32;
1280
1281 return TargetLowering::getRegisterTypeForCallingConv(Context, CC, VT);
1282}
1283
1285 CallingConv::ID CC,
1286 EVT VT) const {
1288 return TargetLowering::getNumRegistersForCallingConv(Context, CC, VT);
1289
1290 if (VT.isVector()) {
1291 unsigned NumElts = VT.getVectorNumElements();
1292 EVT ScalarVT = VT.getScalarType();
1293 unsigned Size = ScalarVT.getSizeInBits();
1294
1295 // FIXME: Should probably promote 8-bit vectors to i16.
1296 if (Size == 16)
1297 return (NumElts + 1) / 2;
1298
1299 if (Size <= 32)
1300 return NumElts;
1301
1302 if (Size > 32)
1303 return NumElts * ((Size + 31) / 32);
1304 } else if (VT.getSizeInBits() > 32)
1305 return (VT.getSizeInBits() + 31) / 32;
1306
1307 return TargetLowering::getNumRegistersForCallingConv(Context, CC, VT);
1308}
1309
1311 LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT,
1312 unsigned &NumIntermediates, MVT &RegisterVT) const {
1313 if (CC != CallingConv::AMDGPU_KERNEL && VT.isVector()) {
1314 unsigned NumElts = VT.getVectorNumElements();
1315 EVT ScalarVT = VT.getScalarType();
1316 unsigned Size = ScalarVT.getSizeInBits();
1317 // FIXME: We should fix the ABI to be the same on targets without 16-bit
1318 // support, but unless we can properly handle 3-vectors, it will be still be
1319 // inconsistent.
1320 if (Size == 16) {
1321 MVT SimpleIntermediateVT =
1323 IntermediateVT = SimpleIntermediateVT;
1324 RegisterVT = Subtarget->has16BitInsts() ? SimpleIntermediateVT : MVT::i32;
1325 NumIntermediates = (NumElts + 1) / 2;
1326 return (NumElts + 1) / 2;
1327 }
1328
1329 if (Size == 32) {
1330 RegisterVT = ScalarVT.getSimpleVT();
1331 IntermediateVT = RegisterVT;
1332 NumIntermediates = NumElts;
1333 return NumIntermediates;
1334 }
1335
1336 if (Size < 16 && Subtarget->has16BitInsts()) {
1337 // FIXME: Should probably form v2i16 pieces
1338 RegisterVT = MVT::i16;
1339 IntermediateVT = ScalarVT;
1340 NumIntermediates = NumElts;
1341 return NumIntermediates;
1342 }
1343
1344 if (Size != 16 && Size <= 32) {
1345 RegisterVT = MVT::i32;
1346 IntermediateVT = ScalarVT;
1347 NumIntermediates = NumElts;
1348 return NumIntermediates;
1349 }
1350
1351 if (Size > 32) {
1352 RegisterVT = MVT::i32;
1353 IntermediateVT = RegisterVT;
1354 NumIntermediates = NumElts * ((Size + 31) / 32);
1355 return NumIntermediates;
1356 }
1357 }
1358
1360 Context, CC, VT, IntermediateVT, NumIntermediates, RegisterVT);
1361}
1362
1364 const DataLayout &DL, Type *Ty,
1365 unsigned MaxNumLanes) {
1366 assert(MaxNumLanes != 0);
1367
1368 LLVMContext &Ctx = Ty->getContext();
1369 if (auto *VT = dyn_cast<FixedVectorType>(Ty)) {
1370 unsigned NumElts = std::min(MaxNumLanes, VT->getNumElements());
1371 return EVT::getVectorVT(Ctx, TLI.getValueType(DL, VT->getElementType()),
1372 NumElts);
1373 }
1374
1375 return TLI.getValueType(DL, Ty);
1376}
1377
1378// Peek through TFE struct returns to only use the data size.
1380 const DataLayout &DL, Type *Ty,
1381 unsigned MaxNumLanes) {
1382 auto *ST = dyn_cast<StructType>(Ty);
1383 if (!ST)
1384 return memVTFromLoadIntrData(TLI, DL, Ty, MaxNumLanes);
1385
1386 // TFE intrinsics return an aggregate type.
1387 assert(ST->getNumContainedTypes() == 2 &&
1388 ST->getContainedType(1)->isIntegerTy(32));
1389 return memVTFromLoadIntrData(TLI, DL, ST->getContainedType(0), MaxNumLanes);
1390}
1391
1392/// Map address space 7 to MVT::amdgpuBufferFatPointer because that's its
1393/// in-memory representation. This return value is a custom type because there
1394/// is no MVT::i160 and adding one breaks integer promotion logic. While this
1395/// could cause issues during codegen, these address space 7 pointers will be
1396/// rewritten away by then. Therefore, we can return MVT::amdgpuBufferFatPointer
1397/// in order to allow pre-codegen passes that query TargetTransformInfo, often
1398/// for cost modeling, to work. (This also sets us up decently for doing the
1399/// buffer lowering in GlobalISel if SelectionDAG ever goes away.)
1401 if (AMDGPUAS::BUFFER_FAT_POINTER == AS && DL.getPointerSizeInBits(AS) == 160)
1402 return MVT::amdgpuBufferFatPointer;
1404 DL.getPointerSizeInBits(AS) == 192)
1405 return MVT::amdgpuBufferStridedPointer;
1407}
1408/// Similarly, the in-memory representation of a p7 is {p8, i32}, aka
1409/// v8i32 when padding is added.
1410/// The in-memory representation of a p9 is {p8, i32, i32}, which is
1411/// also v8i32 with padding.
1413 if ((AMDGPUAS::BUFFER_FAT_POINTER == AS &&
1414 DL.getPointerSizeInBits(AS) == 160) ||
1416 DL.getPointerSizeInBits(AS) == 192))
1417 return MVT::v8i32;
1419}
1420
1421static unsigned getIntrMemWidth(unsigned IntrID) {
1422 switch (IntrID) {
1423 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1424 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1425 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1426 return 8;
1427 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1428 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1429 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1430 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
1431 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
1432 case Intrinsic::amdgcn_flat_load_monitor_b32:
1433 case Intrinsic::amdgcn_global_load_monitor_b32:
1434 return 32;
1435 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1436 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1437 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1438 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
1439 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
1440 case Intrinsic::amdgcn_flat_load_monitor_b64:
1441 case Intrinsic::amdgcn_global_load_monitor_b64:
1442 return 64;
1443 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1444 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128:
1445 case Intrinsic::amdgcn_global_store_async_from_lds_b128:
1446 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B:
1447 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B:
1448 case Intrinsic::amdgcn_flat_load_monitor_b128:
1449 case Intrinsic::amdgcn_global_load_monitor_b128:
1450 return 128;
1451 default:
1452 llvm_unreachable("Unknown width");
1453 }
1454}
1455
1457 unsigned ArgIdx) {
1458 Value *OrderingArg = CI.getArgOperand(ArgIdx);
1459 unsigned Ord = cast<ConstantInt>(OrderingArg)->getZExtValue();
1460 switch (AtomicOrderingCABI(Ord)) {
1463 break;
1466 break;
1469 break;
1470 default:
1472 }
1473}
1474
1475static unsigned parseSyncscopeMDArg(const CallBase &CI, unsigned ArgIdx) {
1476 MDNode *ScopeMD = cast<MDNode>(
1477 cast<MetadataAsValue>(CI.getArgOperand(ArgIdx))->getMetadata());
1478 StringRef Scope = cast<MDString>(ScopeMD->getOperand(0))->getString();
1479 return CI.getContext().getOrInsertSyncScopeID(Scope);
1480}
1481
1483 const CallBase &CI,
1484 MachineFunction &MF,
1485 unsigned IntrID) const {
1487 if (CI.hasMetadata(LLVMContext::MD_invariant_load))
1489 if (CI.hasMetadata(LLVMContext::MD_nontemporal))
1491 Flags |= getTargetMMOFlags(CI);
1492
1493 if (const AMDGPU::RsrcIntrinsic *RsrcIntr =
1495 AttributeSet Attr =
1497 MemoryEffects ME = Attr.getMemoryEffects();
1498 if (ME.doesNotAccessMemory())
1499 return;
1500
1501 bool IsSPrefetch = IntrID == Intrinsic::amdgcn_s_buffer_prefetch_data;
1502 if (!IsSPrefetch) {
1503 auto *Aux = cast<ConstantInt>(CI.getArgOperand(CI.arg_size() - 1));
1504 if (Aux->getZExtValue() & AMDGPU::CPol::VOLATILE)
1506 }
1507
1509
1510 IntrinsicInfo Info;
1511 // TODO: Should images get their own address space?
1513
1514 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode = nullptr;
1515 if (RsrcIntr->IsImage) {
1516 const AMDGPU::ImageDimIntrinsicInfo *Intr =
1518 BaseOpcode = AMDGPU::getMIMGBaseOpcodeInfo(Intr->BaseOpcode);
1519 Info.align.reset();
1520 }
1521
1522 Value *RsrcArg = CI.getArgOperand(RsrcIntr->RsrcArg);
1523 if (auto *RsrcPtrTy = dyn_cast<PointerType>(RsrcArg->getType())) {
1524 if (RsrcPtrTy->getAddressSpace() == AMDGPUAS::BUFFER_RESOURCE)
1525 // We conservatively set the memory operand of a buffer intrinsic to the
1526 // base resource pointer, so that we can access alias information about
1527 // those pointers. Cases like "this points at the same value
1528 // but with a different offset" are handled in
1529 // areMemAccessesTriviallyDisjoint.
1530 Info.ptrVal = RsrcArg;
1531 }
1532
1533 if (ME.onlyReadsMemory()) {
1534 if (RsrcIntr->IsImage) {
1535 unsigned MaxNumLanes = 4;
1536
1537 if (!BaseOpcode->Gather4) {
1538 // If this isn't a gather, we may have excess loaded elements in the
1539 // IR type. Check the dmask for the real number of elements loaded.
1540 unsigned DMask =
1541 cast<ConstantInt>(CI.getArgOperand(0))->getZExtValue();
1542 MaxNumLanes = DMask == 0 ? 1 : llvm::popcount(DMask);
1543 }
1544
1545 Info.memVT = memVTFromLoadIntrReturn(*this, MF.getDataLayout(),
1546 CI.getType(), MaxNumLanes);
1547 } else {
1548 Info.memVT =
1550 std::numeric_limits<unsigned>::max());
1551 }
1552
1553 // FIXME: What does alignment mean for an image?
1554 Info.opc = ISD::INTRINSIC_W_CHAIN;
1555 Info.flags = Flags | MachineMemOperand::MOLoad;
1556 } else if (ME.onlyWritesMemory()) {
1557 Info.opc = ISD::INTRINSIC_VOID;
1558
1559 Type *DataTy = CI.getArgOperand(0)->getType();
1560 if (RsrcIntr->IsImage) {
1561 unsigned DMask = cast<ConstantInt>(CI.getArgOperand(1))->getZExtValue();
1562 unsigned DMaskLanes = DMask == 0 ? 1 : llvm::popcount(DMask);
1563 Info.memVT = memVTFromLoadIntrData(*this, MF.getDataLayout(), DataTy,
1564 DMaskLanes);
1565 } else
1566 Info.memVT = getValueType(MF.getDataLayout(), DataTy);
1567
1568 Info.flags = Flags | MachineMemOperand::MOStore;
1569 } else {
1570 // Atomic, NoReturn Sampler or prefetch
1571 Info.opc = CI.getType()->isVoidTy() ? ISD::INTRINSIC_VOID
1573
1574 switch (IntrID) {
1575 default:
1576 Info.flags = Flags | MachineMemOperand::MOLoad;
1577 if (!IsSPrefetch)
1578 Info.flags |= MachineMemOperand::MOStore;
1579
1580 if ((RsrcIntr->IsImage && BaseOpcode->NoReturn) || IsSPrefetch) {
1581 // Fake memory access type for no return sampler intrinsics
1582 Info.memVT = MVT::i32;
1583 } else {
1584 // XXX - Should this be volatile without known ordering?
1585 Info.flags |= MachineMemOperand::MOVolatile;
1586 Info.memVT = MVT::getVT(CI.getArgOperand(0)->getType());
1587 }
1588 break;
1589 case Intrinsic::amdgcn_raw_buffer_load_lds:
1590 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
1591 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
1592 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
1593 case Intrinsic::amdgcn_struct_buffer_load_lds:
1594 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
1595 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
1596 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
1597 unsigned Width = cast<ConstantInt>(CI.getArgOperand(2))->getZExtValue();
1598
1599 // Entry 0: Load from buffer.
1600 // Don't set an offset, since the pointer value always represents the
1601 // base of the buffer.
1602 Info.memVT = EVT::getIntegerVT(CI.getContext(), Width * 8);
1603 Info.flags = Flags | MachineMemOperand::MOLoad;
1604 Infos.push_back(Info);
1605
1606 // Entry 1: Store to LDS.
1607 // Instruction offset is applied, and an additional per-lane offset
1608 // which we simulate using a larger memory type.
1609 Info.memVT = EVT::getIntegerVT(
1610 CI.getContext(), Width * 8 * Subtarget->getWavefrontSize());
1611 Info.ptrVal = CI.getArgOperand(1); // LDS destination pointer
1612 Info.offset = cast<ConstantInt>(CI.getArgOperand(CI.arg_size() - 2))
1613 ->getZExtValue();
1614 Info.fallbackAddressSpace = AMDGPUAS::LOCAL_ADDRESS;
1615 Info.flags = Flags | MachineMemOperand::MOStore;
1616 Infos.push_back(Info);
1617 return;
1618 }
1619 case Intrinsic::amdgcn_raw_atomic_buffer_load:
1620 case Intrinsic::amdgcn_raw_ptr_atomic_buffer_load:
1621 case Intrinsic::amdgcn_struct_atomic_buffer_load:
1622 case Intrinsic::amdgcn_struct_ptr_atomic_buffer_load: {
1623 Info.memVT =
1625 std::numeric_limits<unsigned>::max());
1626 Info.flags = Flags | MachineMemOperand::MOLoad;
1627 Infos.push_back(Info);
1628 return;
1629 }
1630 }
1631 }
1632 Infos.push_back(Info);
1633 return;
1634 }
1635
1636 IntrinsicInfo Info;
1637 switch (IntrID) {
1638 case Intrinsic::amdgcn_ds_ordered_add:
1639 case Intrinsic::amdgcn_ds_ordered_swap: {
1640 Info.opc = ISD::INTRINSIC_W_CHAIN;
1641 Info.memVT = MVT::getVT(CI.getType());
1642 Info.ptrVal = CI.getOperand(0);
1643 Info.align.reset();
1645
1646 const ConstantInt *Vol = cast<ConstantInt>(CI.getOperand(4));
1647 if (!Vol->isZero())
1648 Info.flags |= MachineMemOperand::MOVolatile;
1649
1650 Infos.push_back(Info);
1651 return;
1652 }
1653 case Intrinsic::amdgcn_ds_add_gs_reg_rtn:
1654 case Intrinsic::amdgcn_ds_sub_gs_reg_rtn: {
1655 Info.opc = ISD::INTRINSIC_W_CHAIN;
1656 Info.memVT = MVT::getVT(CI.getOperand(0)->getType());
1657 Info.ptrVal = nullptr;
1658 Info.fallbackAddressSpace = AMDGPUAS::STREAMOUT_REGISTER;
1660 Infos.push_back(Info);
1661 return;
1662 }
1663 case Intrinsic::amdgcn_ds_append:
1664 case Intrinsic::amdgcn_ds_consume: {
1665 Info.opc = ISD::INTRINSIC_W_CHAIN;
1666 Info.memVT = MVT::getVT(CI.getType());
1667 Info.ptrVal = CI.getOperand(0);
1668 Info.align.reset();
1670
1671 const ConstantInt *Vol = cast<ConstantInt>(CI.getOperand(1));
1672 if (!Vol->isZero())
1673 Info.flags |= MachineMemOperand::MOVolatile;
1674
1675 Infos.push_back(Info);
1676 return;
1677 }
1678 case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
1679 case Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64: {
1680 Info.opc = (IntrID == Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64)
1683 Info.memVT = MVT::getVT(CI.getType());
1684 Info.ptrVal = CI.getOperand(0);
1685 Info.memVT = MVT::i64;
1686 Info.size = 8;
1687 Info.align.reset();
1689 Info.order = AtomicOrdering::Monotonic;
1690 Infos.push_back(Info);
1691 return;
1692 }
1693 case Intrinsic::amdgcn_image_bvh_dual_intersect_ray:
1694 case Intrinsic::amdgcn_image_bvh_intersect_ray:
1695 case Intrinsic::amdgcn_image_bvh8_intersect_ray: {
1696 Info.opc = ISD::INTRINSIC_W_CHAIN;
1697 Info.memVT =
1698 MVT::getVT(IntrID == Intrinsic::amdgcn_image_bvh_intersect_ray
1699 ? CI.getType()
1701 ->getElementType(0)); // XXX: what is correct VT?
1702
1703 Info.fallbackAddressSpace = AMDGPUAS::BUFFER_RESOURCE;
1704 Info.align.reset();
1705 Info.flags = Flags | MachineMemOperand::MOLoad |
1707 Infos.push_back(Info);
1708 return;
1709 }
1710 case Intrinsic::amdgcn_global_atomic_fmin_num:
1711 case Intrinsic::amdgcn_global_atomic_fmax_num:
1712 case Intrinsic::amdgcn_global_atomic_ordered_add_b64:
1713 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1714 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
1715 Info.opc = ISD::INTRINSIC_W_CHAIN;
1716 Info.memVT = MVT::getVT(CI.getType());
1717 Info.ptrVal = CI.getOperand(0);
1718 Info.align.reset();
1719 Info.flags =
1722 Infos.push_back(Info);
1723 return;
1724 }
1725 case Intrinsic::amdgcn_cluster_load_b32:
1726 case Intrinsic::amdgcn_cluster_load_b64:
1727 case Intrinsic::amdgcn_cluster_load_b128:
1728 case Intrinsic::amdgcn_ds_load_tr6_b96:
1729 case Intrinsic::amdgcn_ds_load_tr4_b64:
1730 case Intrinsic::amdgcn_ds_load_tr8_b64:
1731 case Intrinsic::amdgcn_ds_load_tr16_b128:
1732 case Intrinsic::amdgcn_global_load_tr6_b96:
1733 case Intrinsic::amdgcn_global_load_tr4_b64:
1734 case Intrinsic::amdgcn_global_load_tr_b64:
1735 case Intrinsic::amdgcn_global_load_tr_b128:
1736 case Intrinsic::amdgcn_ds_read_tr4_b64:
1737 case Intrinsic::amdgcn_ds_read_tr6_b96:
1738 case Intrinsic::amdgcn_ds_read_tr8_b64:
1739 case Intrinsic::amdgcn_ds_read_tr16_b64: {
1740 Info.opc = ISD::INTRINSIC_W_CHAIN;
1741 Info.memVT = MVT::getVT(CI.getType());
1742 Info.ptrVal = CI.getOperand(0);
1743 Info.align.reset();
1744 Info.flags = Flags | MachineMemOperand::MOLoad;
1745 Infos.push_back(Info);
1746 return;
1747 }
1748 case Intrinsic::amdgcn_flat_load_monitor_b32:
1749 case Intrinsic::amdgcn_flat_load_monitor_b64:
1750 case Intrinsic::amdgcn_flat_load_monitor_b128:
1751 case Intrinsic::amdgcn_global_load_monitor_b32:
1752 case Intrinsic::amdgcn_global_load_monitor_b64:
1753 case Intrinsic::amdgcn_global_load_monitor_b128: {
1754 Info.opc = ISD::INTRINSIC_W_CHAIN;
1755 Info.memVT = EVT::getIntegerVT(CI.getContext(), getIntrMemWidth(IntrID));
1756 Info.ptrVal = CI.getOperand(0);
1757 Info.align.reset();
1758 Info.flags = MachineMemOperand::MOLoad;
1759 Info.order = parseAtomicOrderingCABIArg(CI, 1);
1760 Info.ssid = parseSyncscopeMDArg(CI, 2);
1761 Infos.push_back(Info);
1762 return;
1763 }
1764 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
1765 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
1766 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B: {
1767 Info.opc = ISD::INTRINSIC_W_CHAIN;
1768 Info.memVT = MVT::getVT(CI.getType());
1769 Info.ptrVal = CI.getOperand(0);
1770 Info.align.reset();
1772 Info.order = parseAtomicOrderingCABIArg(CI, 1);
1773 Info.ssid = parseSyncscopeMDArg(CI, 2);
1774 Infos.push_back(Info);
1775 return;
1776 }
1777 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
1778 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
1779 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B: {
1780 Info.opc = ISD::INTRINSIC_VOID;
1781 Info.memVT = MVT::getVT(CI.getArgOperand(1)->getType());
1782 Info.ptrVal = CI.getArgOperand(0);
1783 Info.align.reset();
1785 Info.order = parseAtomicOrderingCABIArg(CI, 2);
1786 Info.ssid = parseSyncscopeMDArg(CI, 3);
1787 Infos.push_back(Info);
1788 return;
1789 }
1790 case Intrinsic::amdgcn_ds_gws_init:
1791 case Intrinsic::amdgcn_ds_gws_barrier:
1792 case Intrinsic::amdgcn_ds_gws_sema_v:
1793 case Intrinsic::amdgcn_ds_gws_sema_br:
1794 case Intrinsic::amdgcn_ds_gws_sema_p:
1795 case Intrinsic::amdgcn_ds_gws_sema_release_all: {
1796 Info.opc = ISD::INTRINSIC_VOID;
1797
1798 const GCNTargetMachine &TM =
1799 static_cast<const GCNTargetMachine &>(getTargetMachine());
1800
1802 Info.ptrVal = MFI->getGWSPSV(TM);
1803
1804 // This is an abstract access, but we need to specify a type and size.
1805 Info.memVT = MVT::i32;
1806 Info.size = 4;
1807 Info.align = Align(4);
1808
1809 if (IntrID == Intrinsic::amdgcn_ds_gws_barrier)
1810 Info.flags = Flags | MachineMemOperand::MOLoad;
1811 else
1812 Info.flags = Flags | MachineMemOperand::MOStore;
1813 Infos.push_back(Info);
1814 return;
1815 }
1816 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1817 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1818 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1819 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1820 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1821 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1822 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1823 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
1824 // Entry 0: Load from source (global/flat).
1825 Info.opc = ISD::INTRINSIC_VOID;
1826 Info.memVT = EVT::getIntegerVT(CI.getContext(), getIntrMemWidth(IntrID));
1827 Info.ptrVal = CI.getArgOperand(0); // Global pointer
1828 Info.offset = cast<ConstantInt>(CI.getArgOperand(2))->getSExtValue();
1829 Info.flags = Flags | MachineMemOperand::MOLoad;
1830 Infos.push_back(Info);
1831
1832 // Entry 1: Store to LDS (same offset).
1833 Info.flags = Flags | MachineMemOperand::MOStore;
1834 Info.ptrVal = CI.getArgOperand(1); // LDS pointer
1835 Infos.push_back(Info);
1836 return;
1837 }
1838 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1839 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1840 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1841 case Intrinsic::amdgcn_global_store_async_from_lds_b128: {
1842 // Entry 0: Load from LDS.
1843 Info.opc = ISD::INTRINSIC_VOID;
1844 Info.memVT = EVT::getIntegerVT(CI.getContext(), getIntrMemWidth(IntrID));
1845 Info.ptrVal = CI.getArgOperand(1); // LDS pointer
1846 Info.offset = cast<ConstantInt>(CI.getArgOperand(2))->getSExtValue();
1847 Info.flags = Flags | MachineMemOperand::MOLoad;
1848 Infos.push_back(Info);
1849
1850 // Entry 1: Store to global (same offset).
1851 Info.flags = Flags | MachineMemOperand::MOStore;
1852 Info.ptrVal = CI.getArgOperand(0); // Global pointer
1853 Infos.push_back(Info);
1854 return;
1855 }
1856 case Intrinsic::amdgcn_av_load_b128:
1857 case Intrinsic::amdgcn_av_store_b128: {
1858 bool IsStore = IntrID == Intrinsic::amdgcn_av_store_b128;
1859 Info.opc = IsStore ? ISD::INTRINSIC_VOID : ISD::INTRINSIC_W_CHAIN;
1860 Info.memVT = MVT::v4i32;
1861 Info.ptrVal = CI.getArgOperand(0);
1862 Info.align = Align(16);
1863 Info.flags |=
1865 // Pretend to be atomic so that SIMemoryLegalizer::expandStore sets cache
1866 // flags appropriately.
1867 Info.order = AtomicOrdering::Monotonic;
1868
1869 LLVMContext &Ctx = CI.getContext();
1870 unsigned ScopeIdx = CI.arg_size() - 1;
1871 MDNode *ScopeMD = cast<MDNode>(
1872 cast<MetadataAsValue>(CI.getArgOperand(ScopeIdx))->getMetadata());
1873 StringRef Scope = cast<MDString>(ScopeMD->getOperand(0))->getString();
1874 Info.ssid = Ctx.getOrInsertSyncScopeID(Scope);
1875 Infos.push_back(Info);
1876 return;
1877 }
1878 case Intrinsic::amdgcn_load_to_lds:
1879 case Intrinsic::amdgcn_load_async_to_lds:
1880 case Intrinsic::amdgcn_global_load_lds:
1881 case Intrinsic::amdgcn_global_load_async_lds: {
1882 unsigned Width = cast<ConstantInt>(CI.getArgOperand(2))->getZExtValue();
1883 auto *Aux = cast<ConstantInt>(CI.getArgOperand(CI.arg_size() - 1));
1884 bool IsVolatile = Aux->getZExtValue() & AMDGPU::CPol::VOLATILE;
1885 if (IsVolatile)
1887
1888 // Entry 0: Load from source (global/flat).
1889 Info.opc = ISD::INTRINSIC_VOID;
1890 Info.memVT = EVT::getIntegerVT(CI.getContext(), Width * 8);
1891 Info.ptrVal = CI.getArgOperand(0); // Source pointer
1892 Info.offset = cast<ConstantInt>(CI.getArgOperand(3))->getSExtValue();
1893 Info.flags = Flags | MachineMemOperand::MOLoad;
1894 Infos.push_back(Info);
1895
1896 // Entry 1: Store to LDS.
1897 // Same offset from the instruction, but an additional per-lane offset is
1898 // added. Represent that using a wider memory type.
1899 Info.memVT = EVT::getIntegerVT(CI.getContext(),
1900 Width * 8 * Subtarget->getWavefrontSize());
1901 Info.ptrVal = CI.getArgOperand(1); // LDS destination pointer
1902 Info.flags = Flags | MachineMemOperand::MOStore;
1903 Infos.push_back(Info);
1904 return;
1905 }
1906 case Intrinsic::amdgcn_ds_bvh_stack_rtn:
1907 case Intrinsic::amdgcn_ds_bvh_stack_push4_pop1_rtn:
1908 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop1_rtn:
1909 case Intrinsic::amdgcn_ds_bvh_stack_push8_pop2_rtn: {
1910 Info.opc = ISD::INTRINSIC_W_CHAIN;
1911
1912 const GCNTargetMachine &TM =
1913 static_cast<const GCNTargetMachine &>(getTargetMachine());
1914
1916 Info.ptrVal = MFI->getGWSPSV(TM);
1917
1918 // This is an abstract access, but we need to specify a type and size.
1919 Info.memVT = MVT::i32;
1920 Info.size = 4;
1921 Info.align = Align(4);
1922
1924 Infos.push_back(Info);
1925 return;
1926 }
1927 case Intrinsic::amdgcn_s_prefetch_data:
1928 case Intrinsic::amdgcn_s_prefetch_inst:
1929 case Intrinsic::amdgcn_flat_prefetch:
1930 case Intrinsic::amdgcn_global_prefetch: {
1931 Info.opc = ISD::INTRINSIC_VOID;
1932 Info.memVT = EVT::getIntegerVT(CI.getContext(), 8);
1933 Info.ptrVal = CI.getArgOperand(0);
1934 Info.flags = Flags | MachineMemOperand::MOLoad;
1935 Infos.push_back(Info);
1936 return;
1937 }
1938 default:
1939 return;
1940 }
1941}
1942
1945 Type *&AccessTy) const {
1946 Value *Ptr = nullptr;
1947 switch (II->getIntrinsicID()) {
1948 case Intrinsic::amdgcn_cluster_load_b128:
1949 case Intrinsic::amdgcn_cluster_load_b64:
1950 case Intrinsic::amdgcn_cluster_load_b32:
1951 case Intrinsic::amdgcn_ds_append:
1952 case Intrinsic::amdgcn_ds_consume:
1953 case Intrinsic::amdgcn_ds_load_tr8_b64:
1954 case Intrinsic::amdgcn_ds_load_tr16_b128:
1955 case Intrinsic::amdgcn_ds_load_tr4_b64:
1956 case Intrinsic::amdgcn_ds_load_tr6_b96:
1957 case Intrinsic::amdgcn_ds_read_tr4_b64:
1958 case Intrinsic::amdgcn_ds_read_tr6_b96:
1959 case Intrinsic::amdgcn_ds_read_tr8_b64:
1960 case Intrinsic::amdgcn_ds_read_tr16_b64:
1961 case Intrinsic::amdgcn_ds_ordered_add:
1962 case Intrinsic::amdgcn_ds_ordered_swap:
1963 case Intrinsic::amdgcn_ds_atomic_async_barrier_arrive_b64:
1964 case Intrinsic::amdgcn_ds_atomic_barrier_arrive_rtn_b64:
1965 case Intrinsic::amdgcn_flat_atomic_fmax_num:
1966 case Intrinsic::amdgcn_flat_atomic_fmin_num:
1967 case Intrinsic::amdgcn_global_atomic_fmax_num:
1968 case Intrinsic::amdgcn_global_atomic_fmin_num:
1969 case Intrinsic::amdgcn_global_atomic_ordered_add_b64:
1970 case Intrinsic::amdgcn_global_load_tr_b64:
1971 case Intrinsic::amdgcn_global_load_tr_b128:
1972 case Intrinsic::amdgcn_global_load_tr4_b64:
1973 case Intrinsic::amdgcn_global_load_tr6_b96:
1974 case Intrinsic::amdgcn_global_store_async_from_lds_b8:
1975 case Intrinsic::amdgcn_global_store_async_from_lds_b32:
1976 case Intrinsic::amdgcn_global_store_async_from_lds_b64:
1977 case Intrinsic::amdgcn_global_store_async_from_lds_b128:
1978 case Intrinsic::amdgcn_av_load_b128:
1979 case Intrinsic::amdgcn_av_store_b128:
1980 Ptr = II->getArgOperand(0);
1981 break;
1982 case Intrinsic::amdgcn_load_to_lds:
1983 case Intrinsic::amdgcn_load_async_to_lds:
1984 case Intrinsic::amdgcn_global_load_lds:
1985 case Intrinsic::amdgcn_global_load_async_lds:
1986 case Intrinsic::amdgcn_global_load_async_to_lds_b8:
1987 case Intrinsic::amdgcn_global_load_async_to_lds_b32:
1988 case Intrinsic::amdgcn_global_load_async_to_lds_b64:
1989 case Intrinsic::amdgcn_global_load_async_to_lds_b128:
1990 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
1991 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
1992 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
1993 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128:
1994 Ptr = II->getArgOperand(1);
1995 break;
1996 default:
1997 return false;
1998 }
1999 AccessTy = II->getType();
2000 Ops.push_back(Ptr);
2001 return true;
2002}
2003
2005 unsigned AddrSpace) const {
2006 if (!Subtarget->hasFlatInstOffsets()) {
2007 // Flat instructions do not have offsets, and only have the register
2008 // address.
2009 return AM.BaseOffs == 0 && AM.Scale == 0;
2010 }
2011
2013 FlatAddrSpace FlatVariant =
2014 AddrSpace == AMDGPUAS::GLOBAL_ADDRESS ? FlatAddrSpace::FlatGlobal
2015 : AddrSpace == AMDGPUAS::PRIVATE_ADDRESS ? FlatAddrSpace::FlatScratch
2016 : FlatAddrSpace::FLAT;
2017
2018 return AM.Scale == 0 &&
2019 (AM.BaseOffs == 0 || Subtarget->getInstrInfo()->isLegalFLATOffset(
2020 AM.BaseOffs, AddrSpace, FlatVariant));
2021}
2022
2024 if (Subtarget->hasFlatGlobalInsts())
2026
2027 if (!Subtarget->hasAddr64() || Subtarget->useFlatForGlobal()) {
2028 // Assume the we will use FLAT for all global memory accesses
2029 // on VI.
2030 // FIXME: This assumption is currently wrong. On VI we still use
2031 // MUBUF instructions for the r + i addressing mode. As currently
2032 // implemented, the MUBUF instructions only work on buffer < 4GB.
2033 // It may be possible to support > 4GB buffers with MUBUF instructions,
2034 // by setting the stride value in the resource descriptor which would
2035 // increase the size limit to (stride * 4GB). However, this is risky,
2036 // because it has never been validated.
2038 }
2039
2040 return isLegalMUBUFAddressingMode(AM);
2041}
2042
2043bool SITargetLowering::isLegalMUBUFAddressingMode(const AddrMode &AM) const {
2044 // MUBUF / MTBUF instructions have a 12-bit unsigned byte offset, and
2045 // additionally can do r + r + i with addr64. 32-bit has more addressing
2046 // mode options. Depending on the resource constant, it can also do
2047 // (i64 r0) + (i32 r1) * (i14 i).
2048 //
2049 // Private arrays end up using a scratch buffer most of the time, so also
2050 // assume those use MUBUF instructions. Scratch loads / stores are currently
2051 // implemented as mubuf instructions with offen bit set, so slightly
2052 // different than the normal addr64.
2053 const SIInstrInfo *TII = Subtarget->getInstrInfo();
2054 if (!TII->isLegalMUBUFImmOffset(AM.BaseOffs))
2055 return false;
2056
2057 // FIXME: Since we can split immediate into soffset and immediate offset,
2058 // would it make sense to allow any immediate?
2059
2060 switch (AM.Scale) {
2061 case 0: // r + i or just i, depending on HasBaseReg.
2062 return true;
2063 case 1:
2064 return true; // We have r + r or r + i.
2065 case 2:
2066 if (AM.HasBaseReg) {
2067 // Reject 2 * r + r.
2068 return false;
2069 }
2070
2071 // Allow 2 * r as r + r
2072 // Or 2 * r + i is allowed as r + r + i.
2073 return true;
2074 default: // Don't allow n * r
2075 return false;
2076 }
2077}
2078
2080 const AddrMode &AM, Type *Ty,
2081 unsigned AS,
2082 Instruction *I) const {
2083 // No global is ever allowed as a base.
2084 if (AM.BaseGV)
2085 return false;
2086
2087 if (AS == AMDGPUAS::GLOBAL_ADDRESS)
2088 return isLegalGlobalAddressingMode(AM);
2089
2090 if (AS == AMDGPUAS::CONSTANT_ADDRESS ||
2094 // If the offset isn't a multiple of 4, it probably isn't going to be
2095 // correctly aligned.
2096 // FIXME: Can we get the real alignment here?
2097 if (AM.BaseOffs % 4 != 0)
2098 return isLegalMUBUFAddressingMode(AM);
2099
2100 if (!Subtarget->hasScalarSubwordLoads()) {
2101 // There are no SMRD extloads, so if we have to do a small type access we
2102 // will use a MUBUF load.
2103 // FIXME?: We also need to do this if unaligned, but we don't know the
2104 // alignment here.
2105 if (Ty->isSized() && DL.getTypeStoreSize(Ty) < 4)
2106 return isLegalGlobalAddressingMode(AM);
2107 }
2108
2109 if (Subtarget->getGeneration() == AMDGPUSubtarget::SOUTHERN_ISLANDS) {
2110 // SMRD instructions have an 8-bit, dword offset on SI.
2111 if (!isUInt<8>(AM.BaseOffs / 4))
2112 return false;
2113 } else if (Subtarget->getGeneration() == AMDGPUSubtarget::SEA_ISLANDS) {
2114 // On CI+, this can also be a 32-bit literal constant offset. If it fits
2115 // in 8-bits, it can use a smaller encoding.
2116 if (!isUInt<32>(AM.BaseOffs / 4))
2117 return false;
2118 } else if (Subtarget->getGeneration() < AMDGPUSubtarget::GFX9) {
2119 // On VI, these use the SMEM format and the offset is 20-bit in bytes.
2120 if (!isUInt<20>(AM.BaseOffs))
2121 return false;
2122 } else if (Subtarget->getGeneration() < AMDGPUSubtarget::GFX12) {
2123 // On GFX9 the offset is signed 21-bit in bytes (but must not be negative
2124 // for S_BUFFER_* instructions).
2125 if (!isInt<21>(AM.BaseOffs))
2126 return false;
2127 } else {
2128 // On GFX12, all offsets are signed 24-bit in bytes.
2129 if (!isInt<24>(AM.BaseOffs))
2130 return false;
2131 }
2132
2133 if ((AS == AMDGPUAS::CONSTANT_ADDRESS ||
2135 AM.BaseOffs < 0) {
2136 // Scalar (non-buffer) loads can only use a negative offset if
2137 // soffset+offset is non-negative. Since the compiler can only prove that
2138 // in a few special cases, it is safer to claim that negative offsets are
2139 // not supported.
2140 return false;
2141 }
2142
2143 if (AM.Scale == 0) // r + i or just i, depending on HasBaseReg.
2144 return true;
2145
2146 if (AM.Scale == 1 && AM.HasBaseReg)
2147 return true;
2148
2149 return false;
2150 }
2151
2152 if (AS == AMDGPUAS::PRIVATE_ADDRESS)
2153 return Subtarget->hasFlatScratchEnabled()
2155 : isLegalMUBUFAddressingMode(AM);
2156
2157 if (AS == AMDGPUAS::LOCAL_ADDRESS ||
2158 (AS == AMDGPUAS::REGION_ADDRESS && Subtarget->hasGDS())) {
2159 // Basic, single offset DS instructions allow a 16-bit unsigned immediate
2160 // field.
2161 // XXX - If doing a 4-byte aligned 8-byte type access, we effectively have
2162 // an 8-bit dword offset but we don't know the alignment here.
2163 if (!isUInt<16>(AM.BaseOffs))
2164 return false;
2165
2166 if (AM.Scale == 0) // r + i or just i, depending on HasBaseReg.
2167 return true;
2168
2169 if (AM.Scale == 1 && AM.HasBaseReg)
2170 return true;
2171
2172 return false;
2173 }
2174
2176 // For an unknown address space, this usually means that this is for some
2177 // reason being used for pure arithmetic, and not based on some addressing
2178 // computation. We don't have instructions that compute pointers with any
2179 // addressing modes, so treat them as having no offset like flat
2180 // instructions.
2182 }
2183
2184 // Assume a user alias of global for unknown address spaces.
2185 return isLegalGlobalAddressingMode(AM);
2186}
2187
2189 const MachineFunction &MF) const {
2191 return (MemVT.getSizeInBits() <= 4 * 32);
2192 if (AS == AMDGPUAS::PRIVATE_ADDRESS) {
2193 unsigned MaxPrivateBits = 8 * getSubtarget()->getMaxPrivateElementSize();
2194 return (MemVT.getSizeInBits() <= MaxPrivateBits);
2195 }
2197 return (MemVT.getSizeInBits() <= 2 * 32);
2198 return true;
2199}
2200
2202 unsigned Size, unsigned AddrSpace, Align Alignment,
2203 MachineMemOperand::Flags Flags, unsigned *IsFast) const {
2204 if (IsFast)
2205 *IsFast = 0;
2206
2207 if (AddrSpace == AMDGPUAS::LOCAL_ADDRESS ||
2208 AddrSpace == AMDGPUAS::REGION_ADDRESS) {
2209 // Check if alignment requirements for ds_read/write instructions are
2210 // disabled.
2211 if (!Subtarget->hasUnalignedDSAccessEnabled() && Alignment < Align(4))
2212 return false;
2213
2214 Align RequiredAlignment(
2215 PowerOf2Ceil(divideCeil(Size, 8))); // Natural alignment.
2216 if (Subtarget->hasLDSMisalignedBugInWGPMode() && Size > 32 &&
2217 Alignment < RequiredAlignment)
2218 return false;
2219
2220 // Either, the alignment requirements are "enabled", or there is an
2221 // unaligned LDS access related hardware bug though alignment requirements
2222 // are "disabled". In either case, we need to check for proper alignment
2223 // requirements.
2224 //
2225 switch (Size) {
2226 case 64:
2227 // SI has a hardware bug in the LDS / GDS bounds checking: if the base
2228 // address is negative, then the instruction is incorrectly treated as
2229 // out-of-bounds even if base + offsets is in bounds. Split vectorized
2230 // loads here to avoid emitting ds_read2_b32. We may re-combine the
2231 // load later in the SILoadStoreOptimizer.
2232 if (!Subtarget->hasUsableDSOffset() && Alignment < Align(8))
2233 return false;
2234
2235 // 8 byte accessing via ds_read/write_b64 require 8-byte alignment, but we
2236 // can do a 4 byte aligned, 8 byte access in a single operation using
2237 // ds_read2/write2_b32 with adjacent offsets.
2238 RequiredAlignment = Align(4);
2239
2240 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2241 // We will either select ds_read_b64/ds_write_b64 or ds_read2_b32/
2242 // ds_write2_b32 depending on the alignment. In either case with either
2243 // alignment there is no faster way of doing this.
2244
2245 // The numbers returned here and below are not additive, it is a 'speed
2246 // rank'. They are just meant to be compared to decide if a certain way
2247 // of lowering an operation is faster than another. For that purpose
2248 // naturally aligned operation gets it bitsize to indicate that "it
2249 // operates with a speed comparable to N-bit wide load". With the full
2250 // alignment ds128 is slower than ds96 for example. If underaligned it
2251 // is comparable to a speed of a single dword access, which would then
2252 // mean 32 < 128 and it is faster to issue a wide load regardless.
2253 // 1 is simply "slow, don't do it". I.e. comparing an aligned load to a
2254 // wider load which will not be aligned anymore the latter is slower.
2255 if (IsFast)
2256 *IsFast = (Alignment >= RequiredAlignment) ? 64
2257 : (Alignment < Align(4)) ? 32
2258 : 1;
2259 return true;
2260 }
2261
2262 break;
2263 case 96:
2264 if (!Subtarget->hasDS96AndDS128())
2265 return false;
2266
2267 // 12 byte accessing via ds_read/write_b96 require 16-byte alignment on
2268 // gfx8 and older.
2269
2270 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2271 // Naturally aligned access is fastest. However, also report it is Fast
2272 // if memory is aligned less than DWORD. A narrow load or store will be
2273 // be equally slow as a single ds_read_b96/ds_write_b96, but there will
2274 // be more of them, so overall we will pay less penalty issuing a single
2275 // instruction.
2276
2277 // See comment on the values above.
2278 if (IsFast)
2279 *IsFast = (Alignment >= RequiredAlignment) ? 96
2280 : (Alignment < Align(4)) ? 32
2281 : 1;
2282 return true;
2283 }
2284
2285 break;
2286 case 128:
2287 if (!Subtarget->hasDS96AndDS128() || !Subtarget->useDS128())
2288 return false;
2289
2290 // 16 byte accessing via ds_read/write_b128 require 16-byte alignment on
2291 // gfx8 and older, but we can do a 8 byte aligned, 16 byte access in a
2292 // single operation using ds_read2/write2_b64.
2293 RequiredAlignment = Align(8);
2294
2295 if (Subtarget->hasUnalignedDSAccessEnabled()) {
2296 // Naturally aligned access is fastest. However, also report it is Fast
2297 // if memory is aligned less than DWORD. A narrow load or store will be
2298 // be equally slow as a single ds_read_b128/ds_write_b128, but there
2299 // will be more of them, so overall we will pay less penalty issuing a
2300 // single instruction.
2301
2302 // See comment on the values above.
2303 if (IsFast)
2304 *IsFast = (Alignment >= RequiredAlignment) ? 128
2305 : (Alignment < Align(4)) ? 32
2306 : 1;
2307 return true;
2308 }
2309
2310 break;
2311 default:
2312 if (Size > 32)
2313 return false;
2314
2315 break;
2316 }
2317
2318 // See comment on the values above.
2319 // Note that we have a single-dword or sub-dword here, so if underaligned
2320 // it is a slowest possible access, hence returned value is 0.
2321 if (IsFast)
2322 *IsFast = (Alignment >= RequiredAlignment) ? Size : 0;
2323
2324 return Alignment >= RequiredAlignment ||
2325 Subtarget->hasUnalignedDSAccessEnabled();
2326 }
2327
2328 // FIXME: We have to be conservative here and assume that flat operations
2329 // will access scratch. If we had access to the IR function, then we
2330 // could determine if any private memory was used in the function.
2331 if (AddrSpace == AMDGPUAS::PRIVATE_ADDRESS ||
2332 AddrSpace == AMDGPUAS::FLAT_ADDRESS) {
2333 bool AlignedBy4 = Alignment >= Align(4);
2334 if (Subtarget->hasUnalignedScratchAccessEnabled()) {
2335 if (IsFast)
2336 *IsFast = AlignedBy4 ? Size : 1;
2337 return true;
2338 }
2339
2340 if (IsFast)
2341 *IsFast = AlignedBy4;
2342
2343 return AlignedBy4;
2344 }
2345
2346 // So long as they are correct, wide global memory operations perform better
2347 // than multiple smaller memory ops -- even when misaligned
2348 if (AMDGPU::isExtendedGlobalAddrSpace(AddrSpace)) {
2349 if (IsFast)
2350 *IsFast = Size;
2351
2352 return Alignment >= Align(4) ||
2353 Subtarget->hasUnalignedBufferAccessEnabled();
2354 }
2355
2356 // Ensure robust out-of-bounds guarantees for buffer accesses are met when the
2357 // "amdgpu.buffer.oob.mode" module flag has not enabled relaxed untyped-buffer
2358 // OOB semantics. Normally hardware will ensure proper
2359 // out-of-bounds behavior, but in the edge case where an access starts
2360 // out-of-bounds and then enters in-bounds, the entire access would be treated
2361 // as out-of-bounds. Prevent misaligned memory accesses by requiring the
2362 // natural alignment of buffer accesses.
2363 if (AddrSpace == AMDGPUAS::BUFFER_FAT_POINTER ||
2364 AddrSpace == AMDGPUAS::BUFFER_RESOURCE ||
2365 AddrSpace == AMDGPUAS::BUFFER_STRIDED_POINTER) {
2366 if (!Subtarget->hasRelaxedBufferOOBMode() &&
2367 Alignment < Align(PowerOf2Ceil(divideCeil(Size, 8))))
2368 return false;
2369 }
2370
2371 // Smaller than dword value must be aligned.
2372 if (Size < 32)
2373 return false;
2374
2375 // 8.1.6 - For Dword or larger reads or writes, the two LSBs of the
2376 // byte-address are ignored, thus forcing Dword alignment.
2377 // This applies to private, global, and constant memory.
2378 if (IsFast)
2379 *IsFast = 1;
2380
2381 return Size >= 32 && Alignment >= Align(4);
2382}
2383
2385 EVT VT, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags,
2386 unsigned *IsFast) const {
2388 Alignment, Flags, IsFast);
2389}
2390
2392 LLVMContext &Context, const MemOp &Op,
2393 const AttributeList &FuncAttributes) const {
2394 // FIXME: Should account for address space here.
2395
2396 // The default fallback uses the private pointer size as a guess for a type to
2397 // use. Make sure we switch these to 64-bit accesses.
2398
2399 if (Op.size() >= 16 &&
2400 Op.isDstAligned(Align(4))) // XXX: Should only do for global
2401 return MVT::v4i32;
2402
2403 if (Op.size() >= 8 && Op.isDstAligned(Align(4)))
2404 return MVT::v2i32;
2405
2406 // Use the default.
2407 return MVT::Other;
2408}
2409
2411 const MemSDNode *MemNode = cast<MemSDNode>(N);
2412 return MemNode->getMemOperand()->getFlags() & MONoClobber;
2413}
2414
2419
2421 unsigned DestAS) const {
2422 if (SrcAS == AMDGPUAS::FLAT_ADDRESS) {
2423 if (DestAS == AMDGPUAS::PRIVATE_ADDRESS &&
2424 Subtarget->hasGloballyAddressableScratch()) {
2425 // Flat -> private requires subtracting src_flat_scratch_base_lo.
2426 return false;
2427 }
2428
2429 // Flat -> private/local is a simple truncate.
2430 // Flat -> global is no-op
2431 return true;
2432 }
2433
2434 const GCNTargetMachine &TM =
2435 static_cast<const GCNTargetMachine &>(getTargetMachine());
2436 return TM.isNoopAddrSpaceCast(DL, SrcAS, DestAS);
2437}
2438
2446
2448 Type *Ty) const {
2449 // FIXME: Could be smarter if called for vector constants.
2450 return true;
2451}
2452
2455 unsigned Index) const {
2458
2459 // TODO: Add more cases that are cheap.
2460 if (Index == 0)
2463}
2464
2465bool SITargetLowering::isExtractVecEltCheap(EVT VT, unsigned Index) const {
2466 // TODO: This should be more aggressive, particular for 16-bit element
2467 // vectors. However there are some mixed improvements and regressions.
2468 EVT EltTy = VT.getVectorElementType();
2469 unsigned MinAlign = Subtarget->useRealTrue16Insts() ? 16 : 32;
2470 return EltTy.getSizeInBits() % MinAlign == 0;
2471}
2472
2474 if (Subtarget->has16BitInsts() && VT == MVT::i16) {
2475 switch (Op) {
2476 case ISD::LOAD:
2477 case ISD::STORE:
2478 return true;
2479 default:
2480 return false;
2481 }
2482 }
2483
2484 // SimplifySetCC uses this function to determine whether or not it should
2485 // create setcc with i1 operands. We don't have instructions for i1 setcc.
2486 if (VT == MVT::i1 && Op == ISD::SETCC)
2487 return false;
2488
2490}
2491
2493 // Do not convert uniform loads to 16-bit.
2494 // Uniform 16-bit loads are legalized to i16 = trunc (zextload i16->i32)
2495 // to match subword load patterns. Allowing conversion back to a 16-bit
2496 // load would create an infinite loop.
2497 if (Subtarget->hasScalarSubwordLoads() && N->getOpcode() == ISD::LOAD &&
2498 !VT.isVector() && VT.getSizeInBits() == 16) {
2499 auto *Load = dyn_cast<LoadSDNode>(N);
2500 if (Load && isUniformLoad(Load)) {
2501 return false;
2502 }
2503 }
2504
2505 return isTypeDesirableForOp(N->getOpcode(), VT);
2506}
2507
2509 const MachineMemOperand *MMO = Load->getMemOperand();
2510
2511 // FIXME: We ought to able able to take the direct isDivergent result. We
2512 // cannot rely on the MMO for a uniformity check, and should stop using
2513 // it. This is a hack for 2 ways that the IR divergence analysis is superior
2514 // to the DAG divergence: Recognizing shift-of-workitem-id as always
2515 // uniform, and isSingleLaneExecution. These should be handled in the DAG
2516 // version, and then this can be dropped.
2517 if (Load->isDivergent() && !AMDGPU::isUniformMMO(MMO))
2518 return false;
2519
2520 return MMO->getSize().hasValue() &&
2521 Load->getAlign() >=
2522 Align(std::min(MMO->getSize().getValue().getKnownMinValue(),
2523 uint64_t(4))) &&
2524 (MMO->isInvariant() ||
2525 (Load->getAddressSpace() == AMDGPUAS::CONSTANT_ADDRESS ||
2526 Load->getAddressSpace() == AMDGPUAS::CONSTANT_ADDRESS_32BIT) ||
2527 (Load->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS &&
2528 Load->isSimple() && isMemOpHasNoClobberedMemOperand(Load)));
2529}
2530
2533 // This isn't really a constant pool but close enough.
2536 return PtrInfo;
2537}
2538
2539SDValue SITargetLowering::lowerKernArgParameterPtr(SelectionDAG &DAG,
2540 const SDLoc &SL,
2541 SDValue Chain,
2542 uint64_t Offset) const {
2543 const DataLayout &DL = DAG.getDataLayout();
2547
2548 auto [InputPtrReg, RC, ArgTy] =
2549 Info->getPreloadedValue(AMDGPUFunctionArgInfo::KERNARG_SEGMENT_PTR);
2550
2551 // We may not have the kernarg segment argument if we have no kernel
2552 // arguments.
2553 if (!InputPtrReg)
2554 return DAG.getConstant(Offset, SL, PtrVT);
2555
2557 SDValue BasePtr = DAG.getCopyFromReg(
2558 Chain, SL, MRI.getLiveInVirtReg(InputPtrReg->getRegister()), PtrVT);
2559
2560 return DAG.getObjectPtrOffset(SL, BasePtr, TypeSize::getFixed(Offset));
2561}
2562
2563SDValue SITargetLowering::getImplicitArgPtr(SelectionDAG &DAG,
2564 const SDLoc &SL) const {
2567 return lowerKernArgParameterPtr(DAG, SL, DAG.getEntryNode(), Offset);
2568}
2569
2570SDValue SITargetLowering::getLDSKernelId(SelectionDAG &DAG,
2571 const SDLoc &SL) const {
2572
2574 std::optional<uint32_t> KnownSize =
2576 if (KnownSize.has_value())
2577 return DAG.getConstant(*KnownSize, SL, MVT::i32);
2578 return SDValue();
2579}
2580
2581SDValue SITargetLowering::convertArgType(SelectionDAG &DAG, EVT VT, EVT MemVT,
2582 const SDLoc &SL, SDValue Val,
2583 bool Signed,
2584 const ISD::InputArg *Arg) const {
2585 // First, if it is a widened vector, narrow it.
2586 if (VT.isVector() &&
2588 EVT NarrowedVT =
2591 Val = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, NarrowedVT, Val,
2592 DAG.getConstant(0, SL, MVT::i32));
2593 }
2594
2595 // Then convert the vector elements or scalar value.
2596 if (Arg && (Arg->Flags.isSExt() || Arg->Flags.isZExt()) && VT.bitsLT(MemVT)) {
2597 unsigned Opc = Arg->Flags.isZExt() ? ISD::AssertZext : ISD::AssertSext;
2598 Val = DAG.getNode(Opc, SL, MemVT, Val, DAG.getValueType(VT));
2599 }
2600
2601 if (MemVT.isFloatingPoint()) {
2602 if (VT.isFloatingPoint()) {
2603 Val = getFPExtOrFPRound(DAG, Val, SL, VT);
2604 } else {
2605 assert(!MemVT.isVector());
2606 EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits());
2607 SDValue Cast = DAG.getBitcast(IntVT, Val);
2608 Val = DAG.getAnyExtOrTrunc(Cast, SL, VT);
2609 }
2610 } else if (Signed)
2611 Val = DAG.getSExtOrTrunc(Val, SL, VT);
2612 else
2613 Val = DAG.getZExtOrTrunc(Val, SL, VT);
2614
2615 return Val;
2616}
2617
2618SDValue SITargetLowering::lowerKernargMemParameter(
2619 SelectionDAG &DAG, EVT VT, EVT MemVT, const SDLoc &SL, SDValue Chain,
2620 uint64_t Offset, Align Alignment, bool Signed,
2621 const ISD::InputArg *Arg) const {
2622
2623 MachinePointerInfo PtrInfo =
2625
2626 // Try to avoid using an extload by loading earlier than the argument address,
2627 // and extracting the relevant bits. The load should hopefully be merged with
2628 // the previous argument.
2629 if (MemVT.getStoreSize() < 4 && Alignment < 4) {
2630 // TODO: Handle align < 4 and size >= 4 (can happen with packed structs).
2631 int64_t AlignDownOffset = alignDown(Offset, 4);
2632 int64_t OffsetDiff = Offset - AlignDownOffset;
2633
2634 EVT IntVT = MemVT.changeTypeToInteger();
2635
2636 // TODO: If we passed in the base kernel offset we could have a better
2637 // alignment than 4, but we don't really need it.
2638 SDValue Ptr = lowerKernArgParameterPtr(DAG, SL, Chain, AlignDownOffset);
2639 SDValue Load = DAG.getLoad(MVT::i32, SL, Chain, Ptr,
2640 PtrInfo.getWithOffset(AlignDownOffset), Align(4),
2643
2644 SDValue ShiftAmt = DAG.getConstant(OffsetDiff * 8, SL, MVT::i32);
2645 SDValue Extract = DAG.getNode(ISD::SRL, SL, MVT::i32, Load, ShiftAmt);
2646
2647 SDValue ArgVal = DAG.getNode(ISD::TRUNCATE, SL, IntVT, Extract);
2648 ArgVal = DAG.getNode(ISD::BITCAST, SL, MemVT, ArgVal);
2649 ArgVal = convertArgType(DAG, VT, MemVT, SL, ArgVal, Signed, Arg);
2650
2651 return DAG.getMergeValues({ArgVal, Load.getValue(1)}, SL);
2652 }
2653
2654 SDValue Ptr = lowerKernArgParameterPtr(DAG, SL, Chain, Offset);
2655 SDValue Load = DAG.getLoad(
2656 MemVT, SL, Chain, Ptr, PtrInfo.getWithOffset(Offset), Alignment,
2658
2659 SDValue Val = convertArgType(DAG, VT, MemVT, SL, Load, Signed, Arg);
2660 return DAG.getMergeValues({Val, Load.getValue(1)}, SL);
2661}
2662
2663/// Coerce an argument which was passed in a different ABI type to the original
2664/// expected value type.
2665SDValue SITargetLowering::convertABITypeToValueType(SelectionDAG &DAG,
2666 SDValue Val,
2667 CCValAssign &VA,
2668 const SDLoc &SL) const {
2669 EVT ValVT = VA.getValVT();
2670
2671 // If this is an 8 or 16-bit value, it is really passed promoted
2672 // to 32 bits. Insert an assert[sz]ext to capture this, then
2673 // truncate to the right size.
2674 switch (VA.getLocInfo()) {
2675 case CCValAssign::Full:
2676 return Val;
2677 case CCValAssign::BCvt:
2678 return DAG.getNode(ISD::BITCAST, SL, ValVT, Val);
2679 case CCValAssign::SExt:
2680 Val = DAG.getNode(ISD::AssertSext, SL, VA.getLocVT(), Val,
2681 DAG.getValueType(ValVT));
2682 return DAG.getNode(ISD::TRUNCATE, SL, ValVT, Val);
2683 case CCValAssign::ZExt:
2684 Val = DAG.getNode(ISD::AssertZext, SL, VA.getLocVT(), Val,
2685 DAG.getValueType(ValVT));
2686 return DAG.getNode(ISD::TRUNCATE, SL, ValVT, Val);
2687 case CCValAssign::AExt:
2688 return DAG.getNode(ISD::TRUNCATE, SL, ValVT, Val);
2689 default:
2690 llvm_unreachable("Unknown loc info!");
2691 }
2692}
2693
2694SDValue SITargetLowering::lowerStackParameter(SelectionDAG &DAG,
2695 CCValAssign &VA, const SDLoc &SL,
2696 SDValue Chain,
2697 const ISD::InputArg &Arg) const {
2699 MachineFrameInfo &MFI = MF.getFrameInfo();
2700
2701 if (Arg.Flags.isByVal()) {
2702 unsigned Size = Arg.Flags.getByValSize();
2703 int FrameIdx = MFI.CreateFixedObject(Size, VA.getLocMemOffset(), false);
2704 return DAG.getFrameIndex(FrameIdx, MVT::i32);
2705 }
2706
2707 unsigned ArgOffset = VA.getLocMemOffset();
2708 unsigned ArgSize = VA.getValVT().getStoreSize();
2709
2710 int FI = MFI.CreateFixedObject(ArgSize, ArgOffset, true);
2711
2712 // Create load nodes to retrieve arguments from the stack.
2713 SDValue FIN = DAG.getFrameIndex(FI, MVT::i32);
2714
2715 // For NON_EXTLOAD, generic code in getLoad assert(ValVT == MemVT)
2717 MVT MemVT = VA.getValVT();
2718
2719 switch (VA.getLocInfo()) {
2720 default:
2721 break;
2722 case CCValAssign::BCvt:
2723 MemVT = VA.getLocVT();
2724 break;
2725 case CCValAssign::SExt:
2726 ExtType = ISD::SEXTLOAD;
2727 break;
2728 case CCValAssign::ZExt:
2729 ExtType = ISD::ZEXTLOAD;
2730 break;
2731 case CCValAssign::AExt:
2732 ExtType = ISD::EXTLOAD;
2733 break;
2734 }
2735
2736 SDValue ArgValue = DAG.getExtLoad(
2737 ExtType, SL, VA.getLocVT(), Chain, FIN,
2739
2740 SDValue ConvertedVal = convertABITypeToValueType(DAG, ArgValue, VA, SL);
2741 if (ConvertedVal == ArgValue)
2742 return ConvertedVal;
2743
2744 return DAG.getMergeValues({ConvertedVal, ArgValue.getValue(1)}, SL);
2745}
2746
2747SDValue SITargetLowering::lowerWorkGroupId(
2748 SelectionDAG &DAG, const SIMachineFunctionInfo &MFI, EVT VT,
2751 AMDGPUFunctionArgInfo::PreloadedValue ClusterWorkGroupIdPV) const {
2752 if (!Subtarget->hasClusters())
2753 return getPreloadedValue(DAG, MFI, VT, WorkGroupIdPV);
2754
2755 // Clusters are supported. Return the global position in the grid. If clusters
2756 // are enabled, WorkGroupIdPV returns the cluster ID not the workgroup ID.
2757
2758 // WorkGroupIdXYZ = ClusterId == 0 ?
2759 // ClusterIdXYZ :
2760 // ClusterIdXYZ * (ClusterMaxIdXYZ + 1) + ClusterWorkGroupIdXYZ
2761 SDValue ClusterIdXYZ = getPreloadedValue(DAG, MFI, VT, WorkGroupIdPV);
2762 SDLoc SL(ClusterIdXYZ);
2763 SDValue ClusterMaxIdXYZ = getPreloadedValue(DAG, MFI, VT, ClusterMaxIdPV);
2764 SDValue One = DAG.getConstant(1, SL, VT);
2765 SDValue ClusterSizeXYZ = DAG.getNode(ISD::ADD, SL, VT, ClusterMaxIdXYZ, One);
2766 SDValue ClusterWorkGroupIdXYZ =
2767 getPreloadedValue(DAG, MFI, VT, ClusterWorkGroupIdPV);
2768 SDValue GlobalIdXYZ =
2769 DAG.getNode(ISD::ADD, SL, VT, ClusterWorkGroupIdXYZ,
2770 DAG.getNode(ISD::MUL, SL, VT, ClusterIdXYZ, ClusterSizeXYZ));
2771
2772 switch (MFI.getClusterDims().getKind()) {
2775 return GlobalIdXYZ;
2777 return ClusterIdXYZ;
2779 using namespace AMDGPU::Hwreg;
2780 SDValue ClusterIdField =
2781 DAG.getTargetConstant(HwregEncoding::encode(ID_IB_STS2, 6, 4), SL, VT);
2782 SDNode *GetReg =
2783 DAG.getMachineNode(AMDGPU::S_GETREG_B32_const, SL, VT, ClusterIdField);
2784 SDValue ClusterId(GetReg, 0);
2785 SDValue Zero = DAG.getConstant(0, SL, VT);
2786 return DAG.getNode(ISD::SELECT_CC, SL, VT, ClusterId, Zero, ClusterIdXYZ,
2787 GlobalIdXYZ, DAG.getCondCode(ISD::SETEQ));
2788 }
2789 }
2790
2791 llvm_unreachable("nothing should reach here");
2792}
2793
2794SDValue SITargetLowering::getPreloadedValue(
2795 SelectionDAG &DAG, const SIMachineFunctionInfo &MFI, EVT VT,
2797 const ArgDescriptor *Reg = nullptr;
2798 const TargetRegisterClass *RC = nullptr;
2799 LLT Ty;
2800
2802 const ArgDescriptor WorkGroupIDX =
2803 ArgDescriptor::createRegister(AMDGPU::TTMP9);
2804 // If GridZ is not programmed in an entry function then the hardware will set
2805 // it to all zeros, so there is no need to mask the GridY value in the low
2806 // order bits.
2807 const ArgDescriptor WorkGroupIDY = ArgDescriptor::createRegister(
2808 AMDGPU::TTMP7,
2809 AMDGPU::isEntryFunctionCC(CC) && !MFI.hasWorkGroupIDZ() ? ~0u : 0xFFFFu);
2810 const ArgDescriptor WorkGroupIDZ =
2811 ArgDescriptor::createRegister(AMDGPU::TTMP7, 0xFFFF0000u);
2812 const ArgDescriptor ClusterWorkGroupIDX =
2813 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x0000000Fu);
2814 const ArgDescriptor ClusterWorkGroupIDY =
2815 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x000000F0u);
2816 const ArgDescriptor ClusterWorkGroupIDZ =
2817 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x00000F00u);
2818 const ArgDescriptor ClusterWorkGroupMaxIDX =
2819 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x0000F000u);
2820 const ArgDescriptor ClusterWorkGroupMaxIDY =
2821 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x000F0000u);
2822 const ArgDescriptor ClusterWorkGroupMaxIDZ =
2823 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x00F00000u);
2824 const ArgDescriptor ClusterWorkGroupMaxFlatID =
2825 ArgDescriptor::createRegister(AMDGPU::TTMP6, 0x0F000000u);
2826
2827 auto LoadConstant = [&](unsigned N) {
2828 return DAG.getConstant(N, SDLoc(), VT);
2829 };
2830
2831 if (Subtarget->hasArchitectedSGPRs() &&
2833 AMDGPU::ClusterDimsAttr ClusterDims = MFI.getClusterDims();
2834 bool HasFixedDims = ClusterDims.isFixedDims();
2835
2836 switch (PVID) {
2838 Reg = &WorkGroupIDX;
2839 RC = &AMDGPU::SReg_32RegClass;
2840 Ty = LLT::scalar(32);
2841 break;
2843 Reg = &WorkGroupIDY;
2844 RC = &AMDGPU::SReg_32RegClass;
2845 Ty = LLT::scalar(32);
2846 break;
2848 Reg = &WorkGroupIDZ;
2849 RC = &AMDGPU::SReg_32RegClass;
2850 Ty = LLT::scalar(32);
2851 break;
2853 if (HasFixedDims && ClusterDims.getDims()[0] == 1)
2854 return LoadConstant(0);
2855 Reg = &ClusterWorkGroupIDX;
2856 RC = &AMDGPU::SReg_32RegClass;
2857 Ty = LLT::scalar(32);
2858 break;
2860 if (HasFixedDims && ClusterDims.getDims()[1] == 1)
2861 return LoadConstant(0);
2862 Reg = &ClusterWorkGroupIDY;
2863 RC = &AMDGPU::SReg_32RegClass;
2864 Ty = LLT::scalar(32);
2865 break;
2867 if (HasFixedDims && ClusterDims.getDims()[2] == 1)
2868 return LoadConstant(0);
2869 Reg = &ClusterWorkGroupIDZ;
2870 RC = &AMDGPU::SReg_32RegClass;
2871 Ty = LLT::scalar(32);
2872 break;
2874 if (HasFixedDims)
2875 return LoadConstant(ClusterDims.getDims()[0] - 1);
2876 Reg = &ClusterWorkGroupMaxIDX;
2877 RC = &AMDGPU::SReg_32RegClass;
2878 Ty = LLT::scalar(32);
2879 break;
2881 if (HasFixedDims)
2882 return LoadConstant(ClusterDims.getDims()[1] - 1);
2883 Reg = &ClusterWorkGroupMaxIDY;
2884 RC = &AMDGPU::SReg_32RegClass;
2885 Ty = LLT::scalar(32);
2886 break;
2888 if (HasFixedDims)
2889 return LoadConstant(ClusterDims.getDims()[2] - 1);
2890 Reg = &ClusterWorkGroupMaxIDZ;
2891 RC = &AMDGPU::SReg_32RegClass;
2892 Ty = LLT::scalar(32);
2893 break;
2895 Reg = &ClusterWorkGroupMaxFlatID;
2896 RC = &AMDGPU::SReg_32RegClass;
2897 Ty = LLT::scalar(32);
2898 break;
2899 default:
2900 break;
2901 }
2902 }
2903
2904 if (!Reg)
2905 std::tie(Reg, RC, Ty) = MFI.getPreloadedValue(PVID);
2906 if (!Reg) {
2908 // It's possible for a kernarg intrinsic call to appear in a kernel with
2909 // no allocated segment, in which case we do not add the user sgpr
2910 // argument, so just return null.
2911 return DAG.getConstant(0, SDLoc(), VT);
2912 }
2913
2914 // It's undefined behavior if a function marked with the amdgpu-no-*
2915 // attributes uses the corresponding intrinsic.
2916 return DAG.getPOISON(VT);
2917 }
2918
2919 return loadInputValue(DAG, RC, VT, SDLoc(DAG.getEntryNode()), *Reg);
2920}
2921
2923 CallingConv::ID CallConv,
2924 ArrayRef<ISD::InputArg> Ins, BitVector &Skipped,
2925 FunctionType *FType,
2926 SIMachineFunctionInfo *Info) {
2927 for (unsigned I = 0, E = Ins.size(), PSInputNum = 0; I != E; ++I) {
2928 const ISD::InputArg *Arg = &Ins[I];
2929
2930 assert((!Arg->VT.isVector() || Arg->VT.getScalarSizeInBits() == 16) &&
2931 "vector type argument should have been split");
2932
2933 // First check if it's a PS input addr.
2934 if (CallConv == CallingConv::AMDGPU_PS && !Arg->Flags.isInReg() &&
2935 PSInputNum <= 15) {
2936 bool SkipArg = !Arg->Used && !Info->isPSInputAllocated(PSInputNum);
2937
2938 // Inconveniently only the first part of the split is marked as isSplit,
2939 // so skip to the end. We only want to increment PSInputNum once for the
2940 // entire split argument.
2941 if (Arg->Flags.isSplit()) {
2942 while (!Arg->Flags.isSplitEnd()) {
2943 assert((!Arg->VT.isVector() || Arg->VT.getScalarSizeInBits() == 16) &&
2944 "unexpected vector split in ps argument type");
2945 if (!SkipArg)
2946 Splits.push_back(*Arg);
2947 Arg = &Ins[++I];
2948 }
2949 }
2950
2951 if (SkipArg) {
2952 // We can safely skip PS inputs.
2953 Skipped.set(Arg->getOrigArgIndex());
2954 ++PSInputNum;
2955 continue;
2956 }
2957
2958 Info->markPSInputAllocated(PSInputNum);
2959 if (Arg->Used)
2960 Info->markPSInputEnabled(PSInputNum);
2961
2962 ++PSInputNum;
2963 }
2964
2965 Splits.push_back(*Arg);
2966 }
2967}
2968
2969// Allocate special inputs passed in VGPRs.
2971 CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI,
2972 SIMachineFunctionInfo &Info) const {
2973 const LLT I32 = LLT::integer(32);
2974 MachineRegisterInfo &MRI = MF.getRegInfo();
2975
2976 if (Info.hasWorkItemIDX()) {
2977 Register Reg = AMDGPU::VGPR0;
2978 MRI.setType(MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass), I32);
2979
2980 CCInfo.AllocateReg(Reg);
2981 unsigned Mask =
2982 (Subtarget->hasPackedTID() && Info.hasWorkItemIDY()) ? 0x3ff : ~0u;
2983 Info.setWorkItemIDX(ArgDescriptor::createRegister(Reg, Mask));
2984 }
2985
2986 if (Info.hasWorkItemIDY()) {
2987 assert(Info.hasWorkItemIDX());
2988 if (Subtarget->hasPackedTID()) {
2989 Info.setWorkItemIDY(
2990 ArgDescriptor::createRegister(AMDGPU::VGPR0, 0x3ff << 10));
2991 } else {
2992 unsigned Reg = AMDGPU::VGPR1;
2993 MRI.setType(MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass), I32);
2994
2995 CCInfo.AllocateReg(Reg);
2996 Info.setWorkItemIDY(ArgDescriptor::createRegister(Reg));
2997 }
2998 }
2999
3000 if (Info.hasWorkItemIDZ()) {
3001 assert(Info.hasWorkItemIDX() && Info.hasWorkItemIDY());
3002 if (Subtarget->hasPackedTID()) {
3003 Info.setWorkItemIDZ(
3004 ArgDescriptor::createRegister(AMDGPU::VGPR0, 0x3ff << 20));
3005 } else {
3006 unsigned Reg = AMDGPU::VGPR2;
3007 MRI.setType(MF.addLiveIn(Reg, &AMDGPU::VGPR_32RegClass), I32);
3008
3009 CCInfo.AllocateReg(Reg);
3010 Info.setWorkItemIDZ(ArgDescriptor::createRegister(Reg));
3011 }
3012 }
3013}
3014
3016 const TargetRegisterClass *RC,
3017 unsigned NumArgRegs) {
3018 ArrayRef<MCPhysReg> ArgSGPRs = ArrayRef(RC->begin(), 32);
3019 unsigned RegIdx = CCInfo.getFirstUnallocated(ArgSGPRs);
3020 if (RegIdx == ArgSGPRs.size())
3021 report_fatal_error("ran out of SGPRs for arguments");
3022
3023 unsigned Reg = ArgSGPRs[RegIdx];
3024 Reg = CCInfo.AllocateReg(Reg);
3025 assert(Reg != AMDGPU::NoRegister);
3026
3027 MachineFunction &MF = CCInfo.getMachineFunction();
3028 MF.addLiveIn(Reg, RC);
3030}
3031
3032// If this has a fixed position, we still should allocate the register in the
3033// CCInfo state. Technically we could get away with this for values passed
3034// outside of the normal argument range.
3036 const TargetRegisterClass *RC,
3037 MCRegister Reg) {
3038 Reg = CCInfo.AllocateReg(Reg);
3039 assert(Reg != AMDGPU::NoRegister);
3040 MachineFunction &MF = CCInfo.getMachineFunction();
3041 MF.addLiveIn(Reg, RC);
3042}
3043
3044static void allocateSGPR32Input(CCState &CCInfo, ArgDescriptor &Arg) {
3045 if (Arg) {
3046 allocateFixedSGPRInputImpl(CCInfo, &AMDGPU::SGPR_32RegClass,
3047 Arg.getRegister());
3048 } else
3049 Arg = allocateSGPR32InputImpl(CCInfo, &AMDGPU::SGPR_32RegClass, 32);
3050}
3051
3052static void allocateSGPR64Input(CCState &CCInfo, ArgDescriptor &Arg) {
3053 if (Arg) {
3054 allocateFixedSGPRInputImpl(CCInfo, &AMDGPU::SGPR_64RegClass,
3055 Arg.getRegister());
3056 } else
3057 Arg = allocateSGPR32InputImpl(CCInfo, &AMDGPU::SGPR_64RegClass, 16);
3058}
3059
3060/// Allocate implicit function VGPR arguments in fixed registers.
3062 CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI,
3063 SIMachineFunctionInfo &Info) const {
3064 Register Reg = CCInfo.AllocateReg(AMDGPU::VGPR31);
3065 if (!Reg)
3066 report_fatal_error("failed to allocate VGPR for implicit arguments");
3067
3068 const unsigned Mask = 0x3ff;
3069 Info.setWorkItemIDX(ArgDescriptor::createRegister(Reg, Mask));
3070 Info.setWorkItemIDY(ArgDescriptor::createRegister(Reg, Mask << 10));
3071 Info.setWorkItemIDZ(ArgDescriptor::createRegister(Reg, Mask << 20));
3072}
3073
3075 CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI,
3076 SIMachineFunctionInfo &Info) const {
3077 auto &ArgInfo = Info.getArgInfo();
3078 const GCNUserSGPRUsageInfo &UserSGPRInfo = Info.getUserSGPRInfo();
3079
3080 // TODO: Unify handling with private memory pointers.
3081 if (UserSGPRInfo.hasDispatchPtr())
3082 allocateSGPR64Input(CCInfo, ArgInfo.DispatchPtr);
3083
3084 if (UserSGPRInfo.hasQueuePtr())
3085 allocateSGPR64Input(CCInfo, ArgInfo.QueuePtr);
3086
3087 // Implicit arg ptr takes the place of the kernarg segment pointer. This is a
3088 // constant offset from the kernarg segment.
3089 if (Info.hasImplicitArgPtr())
3090 allocateSGPR64Input(CCInfo, ArgInfo.ImplicitArgPtr);
3091
3092 if (UserSGPRInfo.hasDispatchID())
3093 allocateSGPR64Input(CCInfo, ArgInfo.DispatchID);
3094
3095 // flat_scratch_init is not applicable for non-kernel functions.
3096
3097 if (Info.hasWorkGroupIDX())
3098 allocateSGPR32Input(CCInfo, ArgInfo.WorkGroupIDX);
3099
3100 if (Info.hasWorkGroupIDY())
3101 allocateSGPR32Input(CCInfo, ArgInfo.WorkGroupIDY);
3102
3103 if (Info.hasWorkGroupIDZ())
3104 allocateSGPR32Input(CCInfo, ArgInfo.WorkGroupIDZ);
3105
3106 if (Info.hasLDSKernelId())
3107 allocateSGPR32Input(CCInfo, ArgInfo.LDSKernelId);
3108}
3109
3110// Allocate special inputs passed in user SGPRs.
3112 MachineFunction &MF,
3113 const SIRegisterInfo &TRI,
3114 SIMachineFunctionInfo &Info) const {
3115 const GCNUserSGPRUsageInfo &UserSGPRInfo = Info.getUserSGPRInfo();
3116 if (UserSGPRInfo.hasImplicitBufferPtr()) {
3117 Register ImplicitBufferPtrReg = Info.addImplicitBufferPtr(TRI);
3118 MF.addLiveIn(ImplicitBufferPtrReg, &AMDGPU::SGPR_64RegClass);
3119 CCInfo.AllocateReg(ImplicitBufferPtrReg);
3120 }
3121
3122 // FIXME: How should these inputs interact with inreg / custom SGPR inputs?
3123 if (UserSGPRInfo.hasPrivateSegmentBuffer()) {
3124 Register PrivateSegmentBufferReg = Info.addPrivateSegmentBuffer(TRI);
3125 MF.addLiveIn(PrivateSegmentBufferReg, &AMDGPU::SGPR_128RegClass);
3126 CCInfo.AllocateReg(PrivateSegmentBufferReg);
3127 }
3128
3129 if (UserSGPRInfo.hasDispatchPtr()) {
3130 Register DispatchPtrReg = Info.addDispatchPtr(TRI);
3131 MF.addLiveIn(DispatchPtrReg, &AMDGPU::SGPR_64RegClass);
3132 CCInfo.AllocateReg(DispatchPtrReg);
3133 }
3134
3135 if (UserSGPRInfo.hasQueuePtr()) {
3136 Register QueuePtrReg = Info.addQueuePtr(TRI);
3137 MF.addLiveIn(QueuePtrReg, &AMDGPU::SGPR_64RegClass);
3138 CCInfo.AllocateReg(QueuePtrReg);
3139 }
3140
3141 if (UserSGPRInfo.hasKernargSegmentPtr()) {
3142 MachineRegisterInfo &MRI = MF.getRegInfo();
3143 Register InputPtrReg = Info.addKernargSegmentPtr(TRI);
3144 CCInfo.AllocateReg(InputPtrReg);
3145
3146 Register VReg = MF.addLiveIn(InputPtrReg, &AMDGPU::SGPR_64RegClass);
3148 }
3149
3150 if (UserSGPRInfo.hasDispatchID()) {
3151 Register DispatchIDReg = Info.addDispatchID(TRI);
3152 MF.addLiveIn(DispatchIDReg, &AMDGPU::SGPR_64RegClass);
3153 CCInfo.AllocateReg(DispatchIDReg);
3154 }
3155
3156 if (UserSGPRInfo.hasFlatScratchInit() && !getSubtarget()->isAmdPalOS()) {
3157 Register FlatScratchInitReg = Info.addFlatScratchInit(TRI);
3158 MF.addLiveIn(FlatScratchInitReg, &AMDGPU::SGPR_64RegClass);
3159 CCInfo.AllocateReg(FlatScratchInitReg);
3160 }
3161
3162 if (UserSGPRInfo.hasPrivateSegmentSize()) {
3163 Register PrivateSegmentSizeReg = Info.addPrivateSegmentSize(TRI);
3164 MF.addLiveIn(PrivateSegmentSizeReg, &AMDGPU::SGPR_32RegClass);
3165 CCInfo.AllocateReg(PrivateSegmentSizeReg);
3166 }
3167
3168 // TODO: Add GridWorkGroupCount user SGPRs when used. For now with HSA we read
3169 // these from the dispatch pointer.
3170}
3171
3172// Allocate pre-loaded kernel arguemtns. Arguments to be preloading must be
3173// sequential starting from the first argument.
3175 CCState &CCInfo, SmallVectorImpl<CCValAssign> &ArgLocs,
3177 const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const {
3178 Function &F = MF.getFunction();
3179 unsigned LastExplicitArgOffset = Subtarget->getExplicitKernelArgOffset();
3180 GCNUserSGPRUsageInfo &SGPRInfo = Info.getUserSGPRInfo();
3181 bool InPreloadSequence = true;
3182 unsigned InIdx = 0;
3183 bool AlignedForImplictArgs = false;
3184 unsigned ImplicitArgOffset = 0;
3185 for (auto &Arg : F.args()) {
3186 if (!InPreloadSequence || !Arg.hasInRegAttr())
3187 break;
3188
3189 unsigned ArgIdx = Arg.getArgNo();
3190 // Don't preload non-original args or parts not in the current preload
3191 // sequence.
3192 if (InIdx < Ins.size() &&
3193 (!Ins[InIdx].isOrigArg() || Ins[InIdx].getOrigArgIndex() != ArgIdx))
3194 break;
3195
3196 for (; InIdx < Ins.size() && Ins[InIdx].isOrigArg() &&
3197 Ins[InIdx].getOrigArgIndex() == ArgIdx;
3198 InIdx++) {
3199 assert(ArgLocs[ArgIdx].isMemLoc());
3200 auto &ArgLoc = ArgLocs[InIdx];
3201 const Align KernelArgBaseAlign = Align(16);
3202 unsigned ArgOffset = ArgLoc.getLocMemOffset();
3203 Align Alignment = commonAlignment(KernelArgBaseAlign, ArgOffset);
3204 unsigned NumAllocSGPRs =
3205 alignTo(ArgLoc.getLocVT().getFixedSizeInBits(), 32) / 32;
3206
3207 // Fix alignment for hidden arguments.
3208 if (Arg.hasAttribute("amdgpu-hidden-argument")) {
3209 if (!AlignedForImplictArgs) {
3210 ImplicitArgOffset =
3211 alignTo(LastExplicitArgOffset,
3212 Subtarget->getAlignmentForImplicitArgPtr()) -
3213 LastExplicitArgOffset;
3214 AlignedForImplictArgs = true;
3215 }
3216 ArgOffset += ImplicitArgOffset;
3217 }
3218
3219 // Arg is preloaded into the previous SGPR.
3220 if (ArgLoc.getLocVT().getStoreSize() < 4 && Alignment < 4) {
3221 assert(InIdx >= 1 && "No previous SGPR");
3222 Info.getArgInfo().PreloadKernArgs[InIdx].Regs.push_back(
3223 Info.getArgInfo().PreloadKernArgs[InIdx - 1].Regs[0]);
3224 continue;
3225 }
3226
3227 unsigned Padding = ArgOffset - LastExplicitArgOffset;
3228 unsigned PaddingSGPRs = alignTo(Padding, 4) / 4;
3229 // Check for free user SGPRs for preloading.
3230 if (PaddingSGPRs + NumAllocSGPRs > SGPRInfo.getNumFreeUserSGPRs()) {
3231 InPreloadSequence = false;
3232 break;
3233 }
3234
3235 // Preload this argument.
3236 const TargetRegisterClass *RC =
3237 TRI.getSGPRClassForBitWidth(NumAllocSGPRs * 32);
3238 SmallVectorImpl<MCRegister> *PreloadRegs =
3239 Info.addPreloadedKernArg(TRI, RC, NumAllocSGPRs, InIdx, PaddingSGPRs);
3240
3241 if (PreloadRegs->size() > 1)
3242 RC = &AMDGPU::SGPR_32RegClass;
3243 for (auto &Reg : *PreloadRegs) {
3244 assert(Reg);
3245 MF.addLiveIn(Reg, RC);
3246 CCInfo.AllocateReg(Reg);
3247 }
3248
3249 LastExplicitArgOffset = NumAllocSGPRs * 4 + ArgOffset;
3250 }
3251 }
3252}
3253
3255 const SIRegisterInfo &TRI,
3256 SIMachineFunctionInfo &Info) const {
3257 // Always allocate this last since it is a synthetic preload.
3258 if (Info.hasLDSKernelId()) {
3259 Register Reg = Info.addLDSKernelId();
3260 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3261 CCInfo.AllocateReg(Reg);
3262 }
3263}
3264
3265// Allocate special input registers that are initialized per-wave.
3268 CallingConv::ID CallConv,
3269 bool IsShader) const {
3270 bool HasArchitectedSGPRs = Subtarget->hasArchitectedSGPRs();
3271 if (Subtarget->hasUserSGPRInit16BugInWave32() && !IsShader) {
3272 // Note: user SGPRs are handled by the front-end for graphics shaders
3273 // Pad up the used user SGPRs with dead inputs.
3274
3275 // TODO: NumRequiredSystemSGPRs computation should be adjusted appropriately
3276 // before enabling architected SGPRs for workgroup IDs.
3277 assert(!HasArchitectedSGPRs && "Unhandled feature for the subtarget");
3278
3279 unsigned CurrentUserSGPRs = Info.getNumUserSGPRs();
3280 // Note we do not count the PrivateSegmentWaveByteOffset. We do not want to
3281 // rely on it to reach 16 since if we end up having no stack usage, it will
3282 // not really be added.
3283 unsigned NumRequiredSystemSGPRs =
3284 Info.hasWorkGroupIDX() + Info.hasWorkGroupIDY() +
3285 Info.hasWorkGroupIDZ() + Info.hasWorkGroupInfo();
3286 for (unsigned i = NumRequiredSystemSGPRs + CurrentUserSGPRs; i < 16; ++i) {
3287 Register Reg = Info.addReservedUserSGPR();
3288 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3289 CCInfo.AllocateReg(Reg);
3290 }
3291 }
3292
3293 if (!HasArchitectedSGPRs) {
3294 if (Info.hasWorkGroupIDX()) {
3295 Register Reg = Info.addWorkGroupIDX();
3296 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3297 CCInfo.AllocateReg(Reg);
3298 }
3299
3300 if (Info.hasWorkGroupIDY()) {
3301 Register Reg = Info.addWorkGroupIDY();
3302 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3303 CCInfo.AllocateReg(Reg);
3304 }
3305
3306 if (Info.hasWorkGroupIDZ()) {
3307 Register Reg = Info.addWorkGroupIDZ();
3308 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3309 CCInfo.AllocateReg(Reg);
3310 }
3311 }
3312
3313 if (Info.hasWorkGroupInfo()) {
3314 Register Reg = Info.addWorkGroupInfo();
3315 MF.addLiveIn(Reg, &AMDGPU::SGPR_32RegClass);
3316 CCInfo.AllocateReg(Reg);
3317 }
3318
3319 if (Info.hasPrivateSegmentWaveByteOffset()) {
3320 // Scratch wave offset passed in system SGPR.
3321 unsigned PrivateSegmentWaveByteOffsetReg;
3322
3323 if (IsShader) {
3324 PrivateSegmentWaveByteOffsetReg =
3325 Info.getPrivateSegmentWaveByteOffsetSystemSGPR();
3326
3327 // This is true if the scratch wave byte offset doesn't have a fixed
3328 // location.
3329 if (PrivateSegmentWaveByteOffsetReg == AMDGPU::NoRegister) {
3330 PrivateSegmentWaveByteOffsetReg = findFirstFreeSGPR(CCInfo);
3331 Info.setPrivateSegmentWaveByteOffset(PrivateSegmentWaveByteOffsetReg);
3332 }
3333 } else
3334 PrivateSegmentWaveByteOffsetReg = Info.addPrivateSegmentWaveByteOffset();
3335
3336 MF.addLiveIn(PrivateSegmentWaveByteOffsetReg, &AMDGPU::SGPR_32RegClass);
3337 CCInfo.AllocateReg(PrivateSegmentWaveByteOffsetReg);
3338 }
3339
3340 assert(!Subtarget->hasUserSGPRInit16BugInWave32() || IsShader ||
3341 Info.getNumPreloadedSGPRs() >= 16);
3342}
3343
3345 MachineFunction &MF,
3346 const SIRegisterInfo &TRI,
3347 SIMachineFunctionInfo &Info) {
3348 // Now that we've figured out where the scratch register inputs are, see if
3349 // should reserve the arguments and use them directly.
3350 MachineFrameInfo &MFI = MF.getFrameInfo();
3351 bool HasStackObjects = MFI.hasStackObjects();
3352 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
3353
3354 // Record that we know we have non-spill stack objects so we don't need to
3355 // check all stack objects later.
3356 if (HasStackObjects)
3357 Info.setHasNonSpillStackObjects(true);
3358
3359 // Everything live out of a block is spilled with fast regalloc, so it's
3360 // almost certain that spilling will be required.
3362 HasStackObjects = true;
3363
3364 // For now assume stack access is needed in any callee functions, so we need
3365 // the scratch registers to pass in.
3366 bool RequiresStackAccess = HasStackObjects || MFI.hasCalls();
3367
3368 if (!ST.hasFlatScratchEnabled()) {
3369 if (RequiresStackAccess && ST.isAmdHsaOrMesa(MF.getFunction())) {
3370 // If we have stack objects, we unquestionably need the private buffer
3371 // resource. For the Code Object V2 ABI, this will be the first 4 user
3372 // SGPR inputs. We can reserve those and use them directly.
3373
3374 Register PrivateSegmentBufferReg =
3376 Info.setScratchRSrcReg(PrivateSegmentBufferReg);
3377 } else {
3378 unsigned ReservedBufferReg = TRI.reservedPrivateSegmentBufferReg(MF);
3379 // We tentatively reserve the last registers (skipping the last registers
3380 // which may contain VCC, FLAT_SCR, and XNACK). After register allocation,
3381 // we'll replace these with the ones immediately after those which were
3382 // really allocated. In the prologue copies will be inserted from the
3383 // argument to these reserved registers.
3384
3385 // Without HSA, relocations are used for the scratch pointer and the
3386 // buffer resource setup is always inserted in the prologue. Scratch wave
3387 // offset is still in an input SGPR.
3388 Info.setScratchRSrcReg(ReservedBufferReg);
3389 }
3390 }
3391
3392 MachineRegisterInfo &MRI = MF.getRegInfo();
3393
3394 // For entry functions we have to set up the stack pointer if we use it,
3395 // whereas non-entry functions get this "for free". This means there is no
3396 // intrinsic advantage to using S32 over S34 in cases where we do not have
3397 // calls but do need a frame pointer (i.e. if we are requested to have one
3398 // because frame pointer elimination is disabled). To keep things simple we
3399 // only ever use S32 as the call ABI stack pointer, and so using it does not
3400 // imply we need a separate frame pointer.
3401 //
3402 // Try to use s32 as the SP, but move it if it would interfere with input
3403 // arguments. This won't work with calls though.
3404 //
3405 // FIXME: Move SP to avoid any possible inputs, or find a way to spill input
3406 // registers.
3407 if (!MRI.isLiveIn(AMDGPU::SGPR32)) {
3408 Info.setStackPtrOffsetReg(AMDGPU::SGPR32);
3409 } else {
3411
3412 if (MFI.hasCalls())
3413 report_fatal_error("call in graphics shader with too many input SGPRs");
3414
3415 for (unsigned Reg : AMDGPU::SGPR_32RegClass) {
3416 if (!MRI.isLiveIn(Reg)) {
3417 Info.setStackPtrOffsetReg(Reg);
3418 break;
3419 }
3420 }
3421
3422 if (Info.getStackPtrOffsetReg() == AMDGPU::SP_REG)
3423 report_fatal_error("failed to find register for SP");
3424 }
3425
3426 // hasFP should be accurate for entry functions even before the frame is
3427 // finalized, because it does not rely on the known stack size, only
3428 // properties like whether variable sized objects are present.
3429 if (ST.getFrameLowering()->hasFP(MF)) {
3430 Info.setFrameOffsetReg(AMDGPU::SGPR33);
3431 }
3432}
3433
3436 return !Info->isEntryFunction();
3437}
3438
3440
3442 MachineBasicBlock *Entry,
3443 const SmallVectorImpl<MachineBasicBlock *> &Exits) const {
3445
3446 const MCPhysReg *IStart = TRI->getCalleeSavedRegsViaCopy(Entry->getParent());
3447 if (!IStart)
3448 return;
3449
3450 const TargetInstrInfo *TII = Subtarget->getInstrInfo();
3451 MachineRegisterInfo *MRI = &Entry->getParent()->getRegInfo();
3452 MachineBasicBlock::iterator MBBI = Entry->begin();
3453 for (const MCPhysReg *I = IStart; *I; ++I) {
3454 const TargetRegisterClass *RC = nullptr;
3455 if (AMDGPU::SReg_64RegClass.contains(*I))
3456 RC = &AMDGPU::SGPR_64RegClass;
3457 else if (AMDGPU::SReg_32RegClass.contains(*I))
3458 RC = &AMDGPU::SGPR_32RegClass;
3459 else
3460 llvm_unreachable("Unexpected register class in CSRsViaCopy!");
3461
3462 Register NewVR = MRI->createVirtualRegister(RC);
3463 // Create copy from CSR to a virtual register.
3464 Entry->addLiveIn(*I);
3465 BuildMI(*Entry, MBBI, DebugLoc(), TII->get(TargetOpcode::COPY), NewVR)
3466 .addReg(*I);
3467
3468 // Insert the copy-back instructions right before the terminator.
3469 for (auto *Exit : Exits)
3470 BuildMI(*Exit, Exit->getFirstTerminator(), DebugLoc(),
3471 TII->get(TargetOpcode::COPY), *I)
3472 .addReg(NewVR);
3473 }
3474}
3475
3477 SDValue Chain, CallingConv::ID CallConv, bool isVarArg,
3478 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
3479 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
3481
3483 const Function &Fn = MF.getFunction();
3486 bool IsError = false;
3487
3488 if (Subtarget->isAmdHsaOS() && AMDGPU::isGraphics(CallConv)) {
3490 Fn, "unsupported non-compute shaders with HSA", DL.getDebugLoc()));
3491 IsError = true;
3492 }
3493
3496 BitVector Skipped(Fn.arg_size());
3497 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), ArgLocs,
3498 *DAG.getContext());
3499
3500 bool IsGraphics = AMDGPU::isGraphics(CallConv);
3501 bool IsKernel = AMDGPU::isKernel(CallConv);
3502 bool IsEntryFunc = AMDGPU::isEntryFunctionCC(CallConv);
3503
3504 if (IsGraphics) {
3505 const GCNUserSGPRUsageInfo &UserSGPRInfo = Info->getUserSGPRInfo();
3506 assert(!UserSGPRInfo.hasDispatchPtr() &&
3507 !UserSGPRInfo.hasKernargSegmentPtr() && !Info->hasWorkGroupInfo() &&
3508 !Info->hasLDSKernelId() && !Info->hasWorkItemIDX() &&
3509 !Info->hasWorkItemIDY() && !Info->hasWorkItemIDZ());
3510 (void)UserSGPRInfo;
3511 if (!Subtarget->hasFlatScratchEnabled())
3512 assert(!UserSGPRInfo.hasFlatScratchInit());
3513 if ((CallConv != CallingConv::AMDGPU_CS &&
3514 CallConv != CallingConv::AMDGPU_Gfx &&
3515 CallConv != CallingConv::AMDGPU_Gfx_WholeWave) ||
3516 !Subtarget->hasArchitectedSGPRs())
3517 assert(!Info->hasWorkGroupIDX() && !Info->hasWorkGroupIDY() &&
3518 !Info->hasWorkGroupIDZ());
3519 }
3520
3521 bool IsWholeWaveFunc = Info->isWholeWaveFunction();
3522
3523 if (CallConv == CallingConv::AMDGPU_PS) {
3524 processPSInputArgs(Splits, CallConv, Ins, Skipped, FType, Info);
3525
3526 // At least one interpolation mode must be enabled or else the GPU will
3527 // hang.
3528 //
3529 // Check PSInputAddr instead of PSInputEnable. The idea is that if the user
3530 // set PSInputAddr, the user wants to enable some bits after the compilation
3531 // based on run-time states. Since we can't know what the final PSInputEna
3532 // will look like, so we shouldn't do anything here and the user should take
3533 // responsibility for the correct programming.
3534 //
3535 // Otherwise, the following restrictions apply:
3536 // - At least one of PERSP_* (0xF) or LINEAR_* (0x70) must be enabled.
3537 // - If POS_W_FLOAT (11) is enabled, at least one of PERSP_* must be
3538 // enabled too.
3539 if ((Info->getPSInputAddr() & 0x7F) == 0 ||
3540 ((Info->getPSInputAddr() & 0xF) == 0 && Info->isPSInputAllocated(11))) {
3541 CCInfo.AllocateReg(AMDGPU::VGPR0);
3542 CCInfo.AllocateReg(AMDGPU::VGPR1);
3543 Info->markPSInputAllocated(0);
3544 Info->markPSInputEnabled(0);
3545 }
3546 if (Subtarget->isAmdPalOS()) {
3547 // For isAmdPalOS, the user does not enable some bits after compilation
3548 // based on run-time states; the register values being generated here are
3549 // the final ones set in hardware. Therefore we need to apply the
3550 // workaround to PSInputAddr and PSInputEnable together. (The case where
3551 // a bit is set in PSInputAddr but not PSInputEnable is where the
3552 // frontend set up an input arg for a particular interpolation mode, but
3553 // nothing uses that input arg. Really we should have an earlier pass
3554 // that removes such an arg.)
3555 unsigned PsInputBits = Info->getPSInputAddr() & Info->getPSInputEnable();
3556 if ((PsInputBits & 0x7F) == 0 ||
3557 ((PsInputBits & 0xF) == 0 && (PsInputBits >> 11 & 1)))
3558 Info->markPSInputEnabled(llvm::countr_zero(Info->getPSInputAddr()));
3559 }
3560 } else if (IsKernel) {
3561 assert(Info->hasWorkGroupIDX() && Info->hasWorkItemIDX());
3562 } else {
3563 Splits.append(IsWholeWaveFunc ? std::next(Ins.begin()) : Ins.begin(),
3564 Ins.end());
3565 }
3566
3567 if (IsKernel)
3568 analyzeFormalArgumentsCompute(CCInfo, Ins);
3569
3570 if (IsEntryFunc) {
3571 allocateSpecialEntryInputVGPRs(CCInfo, MF, *TRI, *Info);
3572 allocateHSAUserSGPRs(CCInfo, MF, *TRI, *Info);
3573 if (IsKernel && Subtarget->hasKernargPreload())
3574 allocatePreloadKernArgSGPRs(CCInfo, ArgLocs, Ins, MF, *TRI, *Info);
3575
3576 allocateLDSKernelId(CCInfo, MF, *TRI, *Info);
3577 } else if (!IsGraphics) {
3578 // For the fixed ABI, pass workitem IDs in the last argument register.
3579 allocateSpecialInputVGPRsFixed(CCInfo, MF, *TRI, *Info);
3580
3581 // FIXME: Sink this into allocateSpecialInputSGPRs
3582 if (!Subtarget->hasFlatScratchEnabled())
3583 CCInfo.AllocateReg(Info->getScratchRSrcReg());
3584
3585 allocateSpecialInputSGPRs(CCInfo, MF, *TRI, *Info);
3586 }
3587
3588 if (!IsKernel) {
3589 CCAssignFn *AssignFn = CCAssignFnForCall(CallConv, isVarArg);
3590 CCInfo.AnalyzeFormalArguments(Splits, AssignFn);
3591
3592 // This assumes the registers are allocated by CCInfo in ascending order
3593 // with no gaps.
3594 Info->setNumWaveDispatchSGPRs(
3595 CCInfo.getFirstUnallocated(AMDGPU::SGPR_32RegClass.getRegisters()));
3596 Info->setNumWaveDispatchVGPRs(
3597 CCInfo.getFirstUnallocated(AMDGPU::VGPR_32RegClass.getRegisters()));
3598 } else if (Info->getNumKernargPreloadedSGPRs()) {
3599 Info->setNumWaveDispatchSGPRs(Info->getNumUserSGPRs());
3600 }
3601
3603
3604 if (IsWholeWaveFunc) {
3605 SDValue Setup = DAG.getNode(AMDGPUISD::WHOLE_WAVE_SETUP, DL,
3606 {MVT::i1, MVT::Other}, Chain);
3607 InVals.push_back(Setup.getValue(0));
3608 Chains.push_back(Setup.getValue(1));
3609 }
3610
3611 // FIXME: This is the minimum kernel argument alignment. We should improve
3612 // this to the maximum alignment of the arguments.
3613 //
3614 // FIXME: Alignment of explicit arguments totally broken with non-0 explicit
3615 // kern arg offset.
3616 const Align KernelArgBaseAlign = Align(16);
3617
3618 for (unsigned i = IsWholeWaveFunc ? 1 : 0, e = Ins.size(), ArgIdx = 0; i != e;
3619 ++i) {
3620 const ISD::InputArg &Arg = Ins[i];
3621 if ((Arg.isOrigArg() && Skipped[Arg.getOrigArgIndex()]) || IsError) {
3622 InVals.push_back(DAG.getPOISON(Arg.VT));
3623 continue;
3624 }
3625
3626 CCValAssign &VA = ArgLocs[ArgIdx++];
3627 MVT VT = VA.getLocVT();
3628
3629 if (IsEntryFunc && VA.isMemLoc()) {
3630 VT = Ins[i].VT;
3631 EVT MemVT = VA.getLocVT();
3632
3633 const uint64_t Offset = VA.getLocMemOffset();
3634 Align Alignment = commonAlignment(KernelArgBaseAlign, Offset);
3635
3636 if (Arg.Flags.isByRef()) {
3637 SDValue Ptr = lowerKernArgParameterPtr(DAG, DL, Chain, Offset);
3638
3639 const GCNTargetMachine &TM =
3640 static_cast<const GCNTargetMachine &>(getTargetMachine());
3641 if (!TM.isNoopAddrSpaceCast(DAG.getDataLayout(),
3643 Arg.Flags.getPointerAddrSpace())) {
3646 }
3647
3648 InVals.push_back(Ptr);
3649 continue;
3650 }
3651
3652 SDValue NewArg;
3653 if (Arg.isOrigArg() && Info->getArgInfo().PreloadKernArgs.count(i)) {
3654 if (MemVT.getStoreSize() < 4 && Alignment < 4) {
3655 // In this case the argument is packed into the previous preload SGPR.
3656 int64_t AlignDownOffset = alignDown(Offset, 4);
3657 int64_t OffsetDiff = Offset - AlignDownOffset;
3658 EVT IntVT = MemVT.changeTypeToInteger();
3659
3660 const SIMachineFunctionInfo *Info =
3663 Register Reg =
3664 Info->getArgInfo().PreloadKernArgs.find(i)->getSecond().Regs[0];
3665
3666 assert(Reg);
3667 Register VReg = MRI.getLiveInVirtReg(Reg);
3668 SDValue Copy = DAG.getCopyFromReg(Chain, DL, VReg, MVT::i32);
3669
3670 SDValue ShiftAmt = DAG.getConstant(OffsetDiff * 8, DL, MVT::i32);
3671 SDValue Extract = DAG.getNode(ISD::SRL, DL, MVT::i32, Copy, ShiftAmt);
3672
3673 SDValue ArgVal = DAG.getNode(ISD::TRUNCATE, DL, IntVT, Extract);
3674 ArgVal = DAG.getNode(ISD::BITCAST, DL, MemVT, ArgVal);
3675 NewArg = convertArgType(DAG, VT, MemVT, DL, ArgVal,
3676 Ins[i].Flags.isSExt(), &Ins[i]);
3677
3678 NewArg = DAG.getMergeValues({NewArg, Copy.getValue(1)}, DL);
3679 } else {
3680 const SIMachineFunctionInfo *Info =
3683 const SmallVectorImpl<MCRegister> &PreloadRegs =
3684 Info->getArgInfo().PreloadKernArgs.find(i)->getSecond().Regs;
3685
3686 SDValue Copy;
3687 if (PreloadRegs.size() == 1) {
3688 Register VReg = MRI.getLiveInVirtReg(PreloadRegs[0]);
3689 const TargetRegisterClass *RC = MRI.getRegClass(VReg);
3690 NewArg = DAG.getCopyFromReg(
3691 Chain, DL, VReg,
3693 TRI->getRegSizeInBits(*RC)));
3694
3695 } else {
3696 // If the kernarg alignment does not match the alignment of the SGPR
3697 // tuple RC that can accommodate this argument, it will be built up
3698 // via copies from from the individual SGPRs that the argument was
3699 // preloaded to.
3701 for (auto Reg : PreloadRegs) {
3702 Register VReg = MRI.getLiveInVirtReg(Reg);
3703 Copy = DAG.getCopyFromReg(Chain, DL, VReg, MVT::i32);
3704 Elts.push_back(Copy);
3705 }
3706 NewArg =
3707 DAG.getBuildVector(EVT::getVectorVT(*DAG.getContext(), MVT::i32,
3708 PreloadRegs.size()),
3709 DL, Elts);
3710 }
3711
3712 // If the argument was preloaded to multiple consecutive 32-bit
3713 // registers because of misalignment between addressable SGPR tuples
3714 // and the argument size, we can still assume that because of kernarg
3715 // segment alignment restrictions that NewArg's size is the same as
3716 // MemVT and just do a bitcast. If MemVT is less than 32-bits we add a
3717 // truncate since we cannot preload to less than a single SGPR and the
3718 // MemVT may be smaller.
3719 EVT MemVTInt =
3721 if (MemVT.bitsLT(NewArg.getSimpleValueType()))
3722 NewArg = DAG.getNode(ISD::TRUNCATE, DL, MemVTInt, NewArg);
3723
3724 NewArg = DAG.getBitcast(MemVT, NewArg);
3725 NewArg = convertArgType(DAG, VT, MemVT, DL, NewArg,
3726 Ins[i].Flags.isSExt(), &Ins[i]);
3727 NewArg = DAG.getMergeValues({NewArg, Chain}, DL);
3728 }
3729 } else {
3730 // Hidden arguments that are in the kernel signature must be preloaded
3731 // to user SGPRs. Print a diagnostic error if a hidden argument is in
3732 // the argument list and is not preloaded.
3733 if (Arg.isOrigArg()) {
3734 Argument *OrigArg = Fn.getArg(Arg.getOrigArgIndex());
3735 if (OrigArg->hasAttribute("amdgpu-hidden-argument")) {
3737 *OrigArg->getParent(),
3738 "hidden argument in kernel signature was not preloaded",
3739 DL.getDebugLoc()));
3740 }
3741 }
3742
3743 NewArg =
3744 lowerKernargMemParameter(DAG, VT, MemVT, DL, Chain, Offset,
3745 Alignment, Ins[i].Flags.isSExt(), &Ins[i]);
3746 }
3747 Chains.push_back(NewArg.getValue(1));
3748
3749 auto *ParamTy =
3750 dyn_cast<PointerType>(FType->getParamType(Ins[i].getOrigArgIndex()));
3751 if (Subtarget->getGeneration() == AMDGPUSubtarget::SOUTHERN_ISLANDS &&
3752 ParamTy &&
3753 (ParamTy->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
3754 ParamTy->getAddressSpace() == AMDGPUAS::REGION_ADDRESS)) {
3755 // On SI local pointers are just offsets into LDS, so they are always
3756 // less than 16-bits. On CI and newer they could potentially be
3757 // real pointers, so we can't guarantee their size.
3758 NewArg = DAG.getNode(ISD::AssertZext, DL, NewArg.getValueType(), NewArg,
3759 DAG.getValueType(MVT::i16));
3760 }
3761
3762 InVals.push_back(NewArg);
3763 continue;
3764 }
3765 if (!IsEntryFunc && VA.isMemLoc()) {
3766 SDValue Val = lowerStackParameter(DAG, VA, DL, Chain, Arg);
3767 InVals.push_back(Val);
3768 if (!Arg.Flags.isByVal())
3769 Chains.push_back(Val.getValue(1));
3770 continue;
3771 }
3772
3773 assert(VA.isRegLoc() && "Parameter must be in a register!");
3774
3775 Register Reg = VA.getLocReg();
3776 const TargetRegisterClass *RC = nullptr;
3777 if (AMDGPU::VGPR_32RegClass.contains(Reg))
3778 RC = &AMDGPU::VGPR_32RegClass;
3779 else if (AMDGPU::SGPR_32RegClass.contains(Reg))
3780 RC = &AMDGPU::SGPR_32RegClass;
3781 else
3782 llvm_unreachable("Unexpected register class in LowerFormalArguments!");
3783
3784 Reg = MF.addLiveIn(Reg, RC);
3785 SDValue Val = DAG.getCopyFromReg(Chain, DL, Reg, VT);
3786 if (Arg.Flags.isInReg() && RC == &AMDGPU::VGPR_32RegClass) {
3787 // FIXME: Need to forward the chains created by `CopyFromReg`s, make sure
3788 // they will read physical regs before any side effect instructions.
3789 SDValue ReadFirstLane =
3790 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
3792 ReadFirstLane, Val);
3793 }
3794
3795 if (Arg.Flags.isSRet()) {
3796 // The return object should be reasonably addressable.
3797 Val = annotateStackObjectPointer(Val, DAG, DL,
3799 }
3800
3801 Val = convertABITypeToValueType(DAG, Val, VA, DL);
3802 InVals.push_back(Val);
3803 }
3804
3805 // Start adding system SGPRs.
3806 if (IsEntryFunc)
3807 allocateSystemSGPRs(CCInfo, MF, *Info, CallConv, IsGraphics);
3808
3809 unsigned StackArgSize = CCInfo.getStackSize();
3810 Info->setBytesInStackArgArea(StackArgSize);
3811
3812 return Chains.empty() ? Chain
3813 : DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
3814}
3815
3816// TODO: If return values can't fit in registers, we should return as many as
3817// possible in registers before passing on stack.
3819 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
3820 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
3821 const Type *RetTy) const {
3822 // Replacing returns with sret/stack usage doesn't make sense for shaders.
3823 // FIXME: Also sort of a workaround for custom vector splitting in LowerReturn
3824 // for shaders. Vector types should be explicitly handled by CC.
3825 if (AMDGPU::isEntryFunctionCC(CallConv))
3826 return true;
3827
3829 CCState CCInfo(CallConv, IsVarArg, MF, RVLocs, Context);
3830 if (!CCInfo.CheckReturn(Outs, CCAssignFnForReturn(CallConv, IsVarArg)))
3831 return false;
3832
3833 // We must use the stack if return would require unavailable registers.
3834 unsigned MaxNumVGPRs = Subtarget->getMaxNumVGPRs(MF);
3835 unsigned TotalNumVGPRs = Subtarget->getAddressableNumArchVGPRs();
3836 for (unsigned i = MaxNumVGPRs; i < TotalNumVGPRs; ++i)
3837 if (CCInfo.isAllocated(AMDGPU::VGPR_32RegClass.getRegister(i)))
3838 return false;
3839
3840 return true;
3841}
3842
3843SDValue
3845 bool isVarArg,
3847 const SmallVectorImpl<SDValue> &OutVals,
3848 const SDLoc &DL, SelectionDAG &DAG) const {
3852
3853 if (AMDGPU::isKernel(CallConv)) {
3854 return AMDGPUTargetLowering::LowerReturn(Chain, CallConv, isVarArg, Outs,
3855 OutVals, DL, DAG);
3856 }
3857
3858 bool IsShader = AMDGPU::isShader(CallConv);
3859
3860 Info->setIfReturnsVoid(Outs.empty());
3861 bool IsWaveEnd = Info->returnsVoid() && IsShader;
3862
3863 // CCValAssign - represent the assignment of the return value to a location.
3865
3866 // CCState - Info about the registers and stack slots.
3867 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs,
3868 *DAG.getContext());
3869
3870 // Analyze outgoing return values.
3871 CCInfo.AnalyzeReturn(Outs, CCAssignFnForReturn(CallConv, isVarArg));
3872
3873 SDValue Glue;
3875 RetOps.push_back(Chain); // Operand #0 = Chain (updated below)
3876
3877 SDValue ReadFirstLane =
3878 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
3879 // Copy the result values into the output registers.
3880 for (unsigned I = 0, RealRVLocIdx = 0, E = RVLocs.size(); I != E;
3881 ++I, ++RealRVLocIdx) {
3882 CCValAssign &VA = RVLocs[I];
3883 assert(VA.isRegLoc() && "Can only return in registers!");
3884 // TODO: Partially return in registers if return values don't fit.
3885 SDValue Arg = OutVals[RealRVLocIdx];
3886
3887 // Copied from other backends.
3888 switch (VA.getLocInfo()) {
3889 case CCValAssign::Full:
3890 break;
3891 case CCValAssign::BCvt:
3892 Arg = DAG.getNode(ISD::BITCAST, DL, VA.getLocVT(), Arg);
3893 break;
3894 case CCValAssign::SExt:
3895 Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg);
3896 break;
3897 case CCValAssign::ZExt:
3898 Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg);
3899 break;
3900 case CCValAssign::AExt:
3901 Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg);
3902 break;
3903 default:
3904 llvm_unreachable("Unknown loc info!");
3905 }
3906 if (TRI->isSGPRPhysReg(VA.getLocReg()))
3908 ReadFirstLane, Arg);
3909 Chain = DAG.getCopyToReg(Chain, DL, VA.getLocReg(), Arg, Glue);
3910 Glue = Chain.getValue(1);
3911 RetOps.push_back(DAG.getRegister(VA.getLocReg(), VA.getLocVT()));
3912 }
3913
3914 // FIXME: Does sret work properly?
3915 if (!Info->isEntryFunction()) {
3916 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
3917 const MCPhysReg *I =
3918 TRI->getCalleeSavedRegsViaCopy(&DAG.getMachineFunction());
3919 if (I) {
3920 for (; *I; ++I) {
3921 if (AMDGPU::SReg_64RegClass.contains(*I))
3922 RetOps.push_back(DAG.getRegister(*I, MVT::i64));
3923 else if (AMDGPU::SReg_32RegClass.contains(*I))
3924 RetOps.push_back(DAG.getRegister(*I, MVT::i32));
3925 else
3926 llvm_unreachable("Unexpected register class in CSRsViaCopy!");
3927 }
3928 }
3929 }
3930
3931 // Update chain and glue.
3932 RetOps[0] = Chain;
3933 if (Glue.getNode())
3934 RetOps.push_back(Glue);
3935
3936 unsigned Opc = AMDGPUISD::ENDPGM;
3937 if (!IsWaveEnd)
3938 Opc = Info->isWholeWaveFunction() ? AMDGPUISD::WHOLE_WAVE_RETURN
3939 : IsShader ? AMDGPUISD::RETURN_TO_EPILOG
3940 : AMDGPUISD::RET_GLUE;
3941 return DAG.getNode(Opc, DL, MVT::Other, RetOps);
3942}
3943
3945 SDValue Chain, SDValue InGlue, CallingConv::ID CallConv, bool IsVarArg,
3946 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
3947 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals, bool IsThisReturn,
3948 SDValue ThisVal) const {
3949 CCAssignFn *RetCC = CCAssignFnForReturn(CallConv, IsVarArg);
3950
3951 // Assign locations to each value returned by this call.
3953 CCState CCInfo(CallConv, IsVarArg, DAG.getMachineFunction(), RVLocs,
3954 *DAG.getContext());
3955 CCInfo.AnalyzeCallResult(Ins, RetCC);
3956
3957 // Copy all of the result registers out of their specified physreg.
3958 for (CCValAssign VA : RVLocs) {
3959 SDValue Val;
3960
3961 if (VA.isRegLoc()) {
3962 Val =
3963 DAG.getCopyFromReg(Chain, DL, VA.getLocReg(), VA.getLocVT(), InGlue);
3964 Chain = Val.getValue(1);
3965 InGlue = Val.getValue(2);
3966 } else if (VA.isMemLoc()) {
3967 report_fatal_error("TODO: return values in memory");
3968 } else
3969 llvm_unreachable("unknown argument location type");
3970
3971 switch (VA.getLocInfo()) {
3972 case CCValAssign::Full:
3973 break;
3974 case CCValAssign::BCvt:
3975 Val = DAG.getNode(ISD::BITCAST, DL, VA.getValVT(), Val);
3976 break;
3977 case CCValAssign::ZExt:
3978 Val = DAG.getNode(ISD::AssertZext, DL, VA.getLocVT(), Val,
3979 DAG.getValueType(VA.getValVT()));
3980 Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val);
3981 break;
3982 case CCValAssign::SExt:
3983 Val = DAG.getNode(ISD::AssertSext, DL, VA.getLocVT(), Val,
3984 DAG.getValueType(VA.getValVT()));
3985 Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val);
3986 break;
3987 case CCValAssign::AExt:
3988 Val = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Val);
3989 break;
3990 default:
3991 llvm_unreachable("Unknown loc info!");
3992 }
3993
3994 InVals.push_back(Val);
3995 }
3996
3997 return Chain;
3998}
3999
4000// Add code to pass special inputs required depending on used features separate
4001// from the explicit user arguments present in the IR.
4003 CallLoweringInfo &CLI, CCState &CCInfo, const SIMachineFunctionInfo &Info,
4004 SmallVectorImpl<std::pair<unsigned, SDValue>> &RegsToPass,
4005 SmallVectorImpl<SDValue> &MemOpChains, SDValue Chain) const {
4006 // If we don't have a call site, this was a call inserted by
4007 // legalization. These can never use special inputs.
4008 if (!CLI.CB)
4009 return;
4010
4011 SelectionDAG &DAG = CLI.DAG;
4012 const SDLoc &DL = CLI.DL;
4013 const Function &F = DAG.getMachineFunction().getFunction();
4014
4015 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
4016 const AMDGPUFunctionArgInfo &CallerArgInfo = Info.getArgInfo();
4017
4018 const AMDGPUFunctionArgInfo &CalleeArgInfo =
4020
4021 // TODO: Unify with private memory register handling. This is complicated by
4022 // the fact that at least in kernels, the input argument is not necessarily
4023 // in the same location as the input.
4024 // clang-format off
4025 static constexpr std::pair<AMDGPUFunctionArgInfo::PreloadedValue,
4026 std::array<StringLiteral, 2>> ImplicitAttrs[] = {
4027 {AMDGPUFunctionArgInfo::DISPATCH_PTR, {"amdgpu-no-dispatch-ptr", ""}},
4028 {AMDGPUFunctionArgInfo::QUEUE_PTR, {"amdgpu-no-queue-ptr", ""}},
4029 {AMDGPUFunctionArgInfo::IMPLICIT_ARG_PTR, {"amdgpu-no-implicitarg-ptr", ""}},
4030 {AMDGPUFunctionArgInfo::DISPATCH_ID, {"amdgpu-no-dispatch-id", ""}},
4031 {AMDGPUFunctionArgInfo::WORKGROUP_ID_X, {"amdgpu-no-workgroup-id-x", "amdgpu-no-cluster-id-x"}},
4032 {AMDGPUFunctionArgInfo::WORKGROUP_ID_Y, {"amdgpu-no-workgroup-id-y", "amdgpu-no-cluster-id-y"}},
4033 {AMDGPUFunctionArgInfo::WORKGROUP_ID_Z, {"amdgpu-no-workgroup-id-z", "amdgpu-no-cluster-id-z"}},
4034 {AMDGPUFunctionArgInfo::LDS_KERNEL_ID, {"amdgpu-no-lds-kernel-id", ""}},
4035 };
4036 // clang-format on
4037
4038 for (auto [InputID, Attrs] : ImplicitAttrs) {
4039 // If the callee does not use the attribute value, skip copying the value.
4040 if (all_of(Attrs, [&](StringRef Attr) {
4041 return Attr.empty() || CLI.CB->hasFnAttr(Attr);
4042 }))
4043 continue;
4044
4045 const auto [OutgoingArg, ArgRC, ArgTy] =
4046 CalleeArgInfo.getPreloadedValue(InputID);
4047 if (!OutgoingArg)
4048 continue;
4049
4050 const auto [IncomingArg, IncomingArgRC, Ty] =
4051 CallerArgInfo.getPreloadedValue(InputID);
4052 assert(IncomingArgRC == ArgRC);
4053
4054 // All special arguments are ints for now.
4055 EVT ArgVT = TRI->getSpillSize(*ArgRC) == 8 ? MVT::i64 : MVT::i32;
4056 SDValue InputReg;
4057
4058 if (IncomingArg) {
4059 InputReg = loadInputValue(DAG, ArgRC, ArgVT, DL, *IncomingArg);
4060 } else if (InputID == AMDGPUFunctionArgInfo::IMPLICIT_ARG_PTR) {
4061 // The implicit arg ptr is special because it doesn't have a corresponding
4062 // input for kernels, and is computed from the kernarg segment pointer.
4063 InputReg = getImplicitArgPtr(DAG, DL);
4064 } else if (InputID == AMDGPUFunctionArgInfo::LDS_KERNEL_ID) {
4065 std::optional<uint32_t> Id =
4067 if (Id.has_value()) {
4068 InputReg = DAG.getConstant(*Id, DL, ArgVT);
4069 } else {
4070 InputReg = DAG.getPOISON(ArgVT);
4071 }
4072 } else {
4073 // We may have proven the input wasn't needed, although the ABI is
4074 // requiring it. We just need to allocate the register appropriately.
4075 InputReg = DAG.getPOISON(ArgVT);
4076 }
4077
4078 if (OutgoingArg->isRegister()) {
4079 RegsToPass.emplace_back(OutgoingArg->getRegister(), InputReg);
4080 if (!CCInfo.AllocateReg(OutgoingArg->getRegister()))
4081 report_fatal_error("failed to allocate implicit input argument");
4082 } else {
4083 unsigned SpecialArgOffset =
4084 CCInfo.AllocateStack(ArgVT.getStoreSize(), Align(4));
4085 SDValue ArgStore =
4086 storeStackInputValue(DAG, DL, Chain, InputReg, SpecialArgOffset);
4087 MemOpChains.push_back(ArgStore);
4088 }
4089 }
4090
4091 // Pack workitem IDs into a single register or pass it as is if already
4092 // packed.
4093
4094 auto [OutgoingArg, ArgRC, Ty] =
4096 if (!OutgoingArg)
4097 std::tie(OutgoingArg, ArgRC, Ty) =
4099 if (!OutgoingArg)
4100 std::tie(OutgoingArg, ArgRC, Ty) =
4102 if (!OutgoingArg)
4103 return;
4104
4105 const ArgDescriptor *IncomingArgX = std::get<0>(
4107 const ArgDescriptor *IncomingArgY = std::get<0>(
4109 const ArgDescriptor *IncomingArgZ = std::get<0>(
4111
4112 SDValue InputReg;
4113 SDLoc SL;
4114
4115 const bool NeedWorkItemIDX = !CLI.CB->hasFnAttr("amdgpu-no-workitem-id-x");
4116 const bool NeedWorkItemIDY = !CLI.CB->hasFnAttr("amdgpu-no-workitem-id-y");
4117 const bool NeedWorkItemIDZ = !CLI.CB->hasFnAttr("amdgpu-no-workitem-id-z");
4118
4119 // If incoming ids are not packed we need to pack them.
4120 if (IncomingArgX && !IncomingArgX->isMasked() && CalleeArgInfo.WorkItemIDX &&
4121 NeedWorkItemIDX) {
4122 if (Subtarget->getMaxWorkitemID(F, 0) != 0) {
4123 InputReg = loadInputValue(DAG, ArgRC, MVT::i32, DL, *IncomingArgX);
4124 } else {
4125 InputReg = DAG.getConstant(0, DL, MVT::i32);
4126 }
4127 }
4128
4129 if (IncomingArgY && !IncomingArgY->isMasked() && CalleeArgInfo.WorkItemIDY &&
4130 NeedWorkItemIDY && Subtarget->getMaxWorkitemID(F, 1) != 0) {
4131 SDValue Y = loadInputValue(DAG, ArgRC, MVT::i32, DL, *IncomingArgY);
4132 Y = DAG.getNode(ISD::SHL, SL, MVT::i32, Y,
4133 DAG.getShiftAmountConstant(10, MVT::i32, SL));
4134 InputReg = InputReg.getNode()
4135 ? DAG.getNode(ISD::OR, SL, MVT::i32, InputReg, Y)
4136 : Y;
4137 }
4138
4139 if (IncomingArgZ && !IncomingArgZ->isMasked() && CalleeArgInfo.WorkItemIDZ &&
4140 NeedWorkItemIDZ && Subtarget->getMaxWorkitemID(F, 2) != 0) {
4141 SDValue Z = loadInputValue(DAG, ArgRC, MVT::i32, DL, *IncomingArgZ);
4142 Z = DAG.getNode(ISD::SHL, SL, MVT::i32, Z,
4143 DAG.getShiftAmountConstant(20, MVT::i32, SL));
4144 InputReg = InputReg.getNode()
4145 ? DAG.getNode(ISD::OR, SL, MVT::i32, InputReg, Z)
4146 : Z;
4147 }
4148
4149 if (!InputReg && (NeedWorkItemIDX || NeedWorkItemIDY || NeedWorkItemIDZ)) {
4150 if (!IncomingArgX && !IncomingArgY && !IncomingArgZ) {
4151 // We're in a situation where the outgoing function requires the workitem
4152 // ID, but the calling function does not have it (e.g a graphics function
4153 // calling a C calling convention function). This is illegal, but we need
4154 // to produce something.
4155 InputReg = DAG.getPOISON(MVT::i32);
4156 } else {
4157 // Workitem ids are already packed, any of present incoming arguments
4158 // will carry all required fields.
4159 ArgDescriptor IncomingArg =
4160 ArgDescriptor::createArg(IncomingArgX ? *IncomingArgX
4161 : IncomingArgY ? *IncomingArgY
4162 : *IncomingArgZ,
4163 ~0u);
4164 InputReg = loadInputValue(DAG, ArgRC, MVT::i32, DL, IncomingArg);
4165 }
4166 }
4167
4168 if (OutgoingArg->isRegister()) {
4169 if (InputReg)
4170 RegsToPass.emplace_back(OutgoingArg->getRegister(), InputReg);
4171
4172 CCInfo.AllocateReg(OutgoingArg->getRegister());
4173 } else {
4174 unsigned SpecialArgOffset = CCInfo.AllocateStack(4, Align(4));
4175 if (InputReg) {
4176 SDValue ArgStore =
4177 storeStackInputValue(DAG, DL, Chain, InputReg, SpecialArgOffset);
4178 MemOpChains.push_back(ArgStore);
4179 }
4180 }
4181}
4182
4184 SDValue Callee, CallingConv::ID CalleeCC, bool IsVarArg,
4186 const SmallVectorImpl<SDValue> &OutVals,
4187 const SmallVectorImpl<ISD::InputArg> &Ins, SelectionDAG &DAG) const {
4188 if (AMDGPU::isChainCC(CalleeCC))
4189 return true;
4190
4191 if (!AMDGPU::mayTailCallThisCC(CalleeCC))
4192 return false;
4193
4194 // For a divergent call target, we need to do a waterfall loop over the
4195 // possible callees which precludes us from using a simple jump.
4196 if (Callee->isDivergent())
4197 return false;
4198
4200 const Function &CallerF = MF.getFunction();
4201 CallingConv::ID CallerCC = CallerF.getCallingConv();
4203 const uint32_t *CallerPreserved = TRI->getCallPreservedMask(MF, CallerCC);
4204
4205 // Kernels aren't callable, and don't have a live in return address so it
4206 // doesn't make sense to do a tail call with entry functions.
4207 if (!CallerPreserved)
4208 return false;
4209
4210 bool CCMatch = CallerCC == CalleeCC;
4211
4213 if (AMDGPU::canGuaranteeTCO(CalleeCC) && CCMatch)
4214 return true;
4215 return false;
4216 }
4217
4218 // TODO: Can we handle var args?
4219 if (IsVarArg)
4220 return false;
4221
4222 for (const Argument &Arg : CallerF.args()) {
4223 if (Arg.hasByValAttr())
4224 return false;
4225 }
4226
4227 LLVMContext &Ctx = *DAG.getContext();
4228
4229 // Check that the call results are passed in the same way.
4230 if (!CCState::resultsCompatible(CalleeCC, CallerCC, MF, Ctx, Ins,
4231 CCAssignFnForCall(CalleeCC, IsVarArg),
4232 CCAssignFnForCall(CallerCC, IsVarArg)))
4233 return false;
4234
4235 // The callee has to preserve all registers the caller needs to preserve.
4236 if (!CCMatch) {
4237 const uint32_t *CalleePreserved = TRI->getCallPreservedMask(MF, CalleeCC);
4238 if (!TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved))
4239 return false;
4240 }
4241
4242 // Nothing more to check if the callee is taking no arguments.
4243 if (Outs.empty())
4244 return true;
4245
4247 CCState CCInfo(CalleeCC, IsVarArg, MF, ArgLocs, Ctx);
4248
4249 // FIXME: We are not allocating special input registers, so we will be
4250 // deciding based on incorrect register assignments.
4251 CCInfo.AnalyzeCallOperands(Outs, CCAssignFnForCall(CalleeCC, IsVarArg));
4252
4253 const SIMachineFunctionInfo *FuncInfo = MF.getInfo<SIMachineFunctionInfo>();
4254 // If the stack arguments for this call do not fit into our own save area then
4255 // the call cannot be made tail.
4256 // TODO: Is this really necessary?
4257 if (CCInfo.getStackSize() > FuncInfo->getBytesInStackArgArea())
4258 return false;
4259
4260 for (const auto &[CCVA, ArgVal] : zip_equal(ArgLocs, OutVals)) {
4261 // FIXME: What about inreg arguments that end up passed in memory?
4262 if (!CCVA.isRegLoc())
4263 continue;
4264
4265 // If we are passing an argument in an SGPR, and the value is divergent,
4266 // this call requires a waterfall loop.
4267 if (ArgVal->isDivergent() && TRI->isSGPRPhysReg(CCVA.getLocReg())) {
4268 LLVM_DEBUG(
4269 dbgs() << "Cannot tail call due to divergent outgoing argument in "
4270 << printReg(CCVA.getLocReg(), TRI) << '\n');
4271 return false;
4272 }
4273 }
4274
4275 const MachineRegisterInfo &MRI = MF.getRegInfo();
4276 return parametersInCSRMatch(MRI, CallerPreserved, ArgLocs, OutVals);
4277}
4278
4280 if (!CI->isTailCall())
4281 return false;
4282
4283 const Function *ParentFn = CI->getFunction();
4285 return false;
4286 return true;
4287}
4288
4289namespace {
4290// Chain calls have special arguments that we need to handle. These are
4291// tagging along at the end of the arguments list(s), after the SGPR and VGPR
4292// arguments (index 0 and 1 respectively).
4293enum ChainCallArgIdx {
4294 Exec = 2,
4295 Flags,
4296 NumVGPRs,
4297 FallbackExec,
4298 FallbackCallee
4299};
4300} // anonymous namespace
4301
4302// The wave scratch offset register is used as the global base pointer.
4304 SmallVectorImpl<SDValue> &InVals) const {
4305 CallingConv::ID CallConv = CLI.CallConv;
4306 bool IsChainCallConv = AMDGPU::isChainCC(CallConv);
4307
4308 SelectionDAG &DAG = CLI.DAG;
4309
4310 const SDLoc &DL = CLI.DL;
4311 SDValue Chain = CLI.Chain;
4312 SDValue Callee = CLI.Callee;
4313
4314 llvm::SmallVector<SDValue, 6> ChainCallSpecialArgs;
4315 bool UsesDynamicVGPRs = false;
4316 if (IsChainCallConv) {
4317 // The last arguments should be the value that we need to put in EXEC,
4318 // followed by the flags and any other arguments with special meanings.
4319 // Pop them out of CLI.Outs and CLI.OutVals before we do any processing so
4320 // we don't treat them like the "real" arguments.
4321 auto RequestedExecIt =
4322 llvm::find_if(CLI.Outs, [](const ISD::OutputArg &Arg) {
4323 return Arg.OrigArgIndex == 2;
4324 });
4325 assert(RequestedExecIt != CLI.Outs.end() && "No node for EXEC");
4326
4327 size_t SpecialArgsBeginIdx = RequestedExecIt - CLI.Outs.begin();
4328 CLI.OutVals.erase(CLI.OutVals.begin() + SpecialArgsBeginIdx,
4329 CLI.OutVals.end());
4330 CLI.Outs.erase(RequestedExecIt, CLI.Outs.end());
4331
4332 assert(CLI.Outs.back().OrigArgIndex < 2 &&
4333 "Haven't popped all the special args");
4334
4335 TargetLowering::ArgListEntry RequestedExecArg =
4336 CLI.Args[ChainCallArgIdx::Exec];
4337 if (!RequestedExecArg.Ty->isIntegerTy(Subtarget->getWavefrontSize()))
4338 return lowerUnhandledCall(CLI, InVals, "Invalid value for EXEC");
4339
4340 // Convert constants into TargetConstants, so they become immediate operands
4341 // instead of being selected into S_MOV.
4342 auto PushNodeOrTargetConstant = [&](TargetLowering::ArgListEntry Arg) {
4343 if (const auto *ArgNode = dyn_cast<ConstantSDNode>(Arg.Node)) {
4344 ChainCallSpecialArgs.push_back(DAG.getTargetConstant(
4345 ArgNode->getAPIntValue(), DL, ArgNode->getValueType(0)));
4346 } else
4347 ChainCallSpecialArgs.push_back(Arg.Node);
4348 };
4349
4350 PushNodeOrTargetConstant(RequestedExecArg);
4351
4352 // Process any other special arguments depending on the value of the flags.
4353 TargetLowering::ArgListEntry Flags = CLI.Args[ChainCallArgIdx::Flags];
4354
4355 const APInt &FlagsValue = cast<ConstantSDNode>(Flags.Node)->getAPIntValue();
4356 if (FlagsValue.isZero()) {
4357 if (CLI.Args.size() > ChainCallArgIdx::Flags + 1)
4358 return lowerUnhandledCall(CLI, InVals,
4359 "no additional args allowed if flags == 0");
4360 } else if (FlagsValue.isOneBitSet(0)) {
4361 if (CLI.Args.size() != ChainCallArgIdx::FallbackCallee + 1) {
4362 return lowerUnhandledCall(CLI, InVals, "expected 3 additional args");
4363 }
4364
4365 if (!Subtarget->isWave32()) {
4366 return lowerUnhandledCall(
4367 CLI, InVals, "dynamic VGPR mode is only supported for wave32");
4368 }
4369
4370 UsesDynamicVGPRs = true;
4371 std::for_each(CLI.Args.begin() + ChainCallArgIdx::NumVGPRs,
4372 CLI.Args.end(), PushNodeOrTargetConstant);
4373 }
4374 }
4375
4377 SmallVector<SDValue, 32> &OutVals = CLI.OutVals;
4379 bool &IsTailCall = CLI.IsTailCall;
4380 bool IsVarArg = CLI.IsVarArg;
4381 bool IsSibCall = false;
4383
4384 if (Callee.isUndef() || isNullConstant(Callee)) {
4385 if (!CLI.IsTailCall) {
4386 for (ISD::InputArg &Arg : CLI.Ins)
4387 InVals.push_back(DAG.getPOISON(Arg.VT));
4388 }
4389
4390 return Chain;
4391 }
4392
4393 if (IsVarArg) {
4394 return lowerUnhandledCall(CLI, InVals,
4395 "unsupported call to variadic function ");
4396 }
4397
4398 if (!CLI.CB)
4399 return lowerUnhandledCall(CLI, InVals, "unsupported libcall legalization");
4400
4401 if (IsTailCall && MF.getTarget().Options.GuaranteedTailCallOpt) {
4402 return lowerUnhandledCall(CLI, InVals,
4403 "unsupported required tail call to function ");
4404 }
4405
4406 if (IsTailCall) {
4407 IsTailCall = isEligibleForTailCallOptimization(Callee, CallConv, IsVarArg,
4408 Outs, OutVals, Ins, DAG);
4409 if (!IsTailCall &&
4410 ((CLI.CB && CLI.CB->isMustTailCall()) || IsChainCallConv)) {
4411 report_fatal_error("failed to perform tail call elimination on a call "
4412 "site marked musttail or on llvm.amdgcn.cs.chain");
4413 }
4414
4415 bool TailCallOpt = MF.getTarget().Options.GuaranteedTailCallOpt;
4416
4417 // A sibling call is one where we're under the usual C ABI and not planning
4418 // to change that but can still do a tail call:
4419 if (!TailCallOpt && IsTailCall)
4420 IsSibCall = true;
4421
4422 if (IsTailCall)
4423 ++NumTailCalls;
4424 }
4425
4428 SmallVector<SDValue, 8> MemOpChains;
4429
4430 // Analyze operands of the call, assigning locations to each operand.
4432 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
4433 CCAssignFn *AssignFn = CCAssignFnForCall(CallConv, IsVarArg);
4434
4435 if (CallConv != CallingConv::AMDGPU_Gfx && !AMDGPU::isChainCC(CallConv) &&
4437 // With a fixed ABI, allocate fixed registers before user arguments.
4438 passSpecialInputs(CLI, CCInfo, *Info, RegsToPass, MemOpChains, Chain);
4439 }
4440
4441 // Mark the scratch resource descriptor as allocated so the CC analysis
4442 // does not assign user arguments to these registers, matching the callee.
4443 if (!Subtarget->hasFlatScratchEnabled())
4444 CCInfo.AllocateReg(Info->getScratchRSrcReg());
4445
4446 CCInfo.AnalyzeCallOperands(Outs, AssignFn);
4447
4448 // Get a count of how many bytes are to be pushed on the stack.
4449 unsigned NumBytes = CCInfo.getStackSize();
4450
4451 if (IsSibCall) {
4452 // Since we're not changing the ABI to make this a tail call, the memory
4453 // operands are already available in the caller's incoming argument space.
4454 NumBytes = 0;
4455 }
4456
4457 // FPDiff is the byte offset of the call's argument area from the callee's.
4458 // Stores to callee stack arguments will be placed in FixedStackSlots offset
4459 // by this amount for a tail call. In a sibling call it must be 0 because the
4460 // caller will deallocate the entire stack and the callee still expects its
4461 // arguments to begin at SP+0. Completely unused for non-tail calls.
4462 int32_t FPDiff = 0;
4463 MachineFrameInfo &MFI = MF.getFrameInfo();
4464 auto *TRI = Subtarget->getRegisterInfo();
4465
4466 // Adjust the stack pointer for the new arguments...
4467 // These operations are automatically eliminated by the prolog/epilog pass
4468 if (!IsSibCall)
4469 Chain = DAG.getCALLSEQ_START(Chain, 0, 0, DL);
4470
4471 if (!IsSibCall || IsChainCallConv) {
4472 if (!Subtarget->hasFlatScratchEnabled()) {
4473 SmallVector<SDValue, 4> CopyFromChains;
4474
4475 // In the HSA case, this should be an identity copy.
4476 SDValue ScratchRSrcReg =
4477 DAG.getCopyFromReg(Chain, DL, Info->getScratchRSrcReg(), MVT::v4i32);
4478 RegsToPass.emplace_back(IsChainCallConv
4479 ? AMDGPU::SGPR48_SGPR49_SGPR50_SGPR51
4480 : AMDGPU::SGPR0_SGPR1_SGPR2_SGPR3,
4481 ScratchRSrcReg);
4482 CopyFromChains.push_back(ScratchRSrcReg.getValue(1));
4483 Chain = DAG.getTokenFactor(DL, CopyFromChains);
4484 }
4485 }
4486
4487 const unsigned NumSpecialInputs = RegsToPass.size();
4488
4489 MVT PtrVT = MVT::i32;
4490
4491 // Walk the register/memloc assignments, inserting copies/loads.
4492 for (unsigned i = 0, e = ArgLocs.size(); i != e; ++i) {
4493 CCValAssign &VA = ArgLocs[i];
4494 SDValue Arg = OutVals[i];
4495
4496 // Promote the value if needed.
4497 switch (VA.getLocInfo()) {
4498 case CCValAssign::Full:
4499 break;
4500 case CCValAssign::BCvt:
4501 Arg = DAG.getNode(ISD::BITCAST, DL, VA.getLocVT(), Arg);
4502 break;
4503 case CCValAssign::ZExt:
4504 Arg = DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Arg);
4505 break;
4506 case CCValAssign::SExt:
4507 Arg = DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Arg);
4508 break;
4509 case CCValAssign::AExt:
4510 Arg = DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Arg);
4511 break;
4512 case CCValAssign::FPExt:
4513 Arg = DAG.getNode(ISD::FP_EXTEND, DL, VA.getLocVT(), Arg);
4514 break;
4515 default:
4516 llvm_unreachable("Unknown loc info!");
4517 }
4518
4519 if (VA.isRegLoc()) {
4520 RegsToPass.push_back(std::pair(VA.getLocReg(), Arg));
4521 } else {
4522 assert(VA.isMemLoc());
4523
4524 SDValue DstAddr;
4525 MachinePointerInfo DstInfo;
4526
4527 unsigned LocMemOffset = VA.getLocMemOffset();
4528 int32_t Offset = LocMemOffset;
4529
4530 SDValue PtrOff = DAG.getConstant(Offset, DL, PtrVT);
4531 MaybeAlign Alignment;
4532
4533 if (IsTailCall) {
4534 ISD::ArgFlagsTy Flags = Outs[i].Flags;
4535 unsigned OpSize = Flags.isByVal() ? Flags.getByValSize()
4536 : VA.getValVT().getStoreSize();
4537
4538 // FIXME: We can have better than the minimum byval required alignment.
4539 Alignment =
4540 Flags.isByVal()
4541 ? Flags.getNonZeroByValAlign()
4542 : commonAlignment(Subtarget->getStackAlignment(), Offset);
4543
4544 Offset = Offset + FPDiff;
4545 int FI = MFI.CreateFixedObject(OpSize, Offset, true);
4546
4547 DstAddr = DAG.getFrameIndex(FI, PtrVT);
4548 DstInfo = MachinePointerInfo::getFixedStack(MF, FI);
4549
4550 // Make sure any stack arguments overlapping with where we're storing
4551 // are loaded before this eventual operation. Otherwise they'll be
4552 // clobbered.
4553
4554 // FIXME: Why is this really necessary? This seems to just result in a
4555 // lot of code to copy the stack and write them back to the same
4556 // locations, which are supposed to be immutable?
4557 Chain = addTokenForArgument(Chain, DAG, MFI, FI);
4558 } else {
4559 // Stores to the argument stack area are relative to the stack pointer.
4560 SDValue SP = DAG.getCopyFromReg(Chain, DL, Info->getStackPtrOffsetReg(),
4561 MVT::i32);
4562 DstAddr = DAG.getNode(ISD::ADD, DL, MVT::i32, SP, PtrOff);
4563 DstInfo = MachinePointerInfo::getStack(MF, LocMemOffset);
4564 Alignment =
4565 commonAlignment(Subtarget->getStackAlignment(), LocMemOffset);
4566 }
4567
4568 if (Outs[i].Flags.isByVal()) {
4569 SDValue SizeNode =
4570 DAG.getConstant(Outs[i].Flags.getByValSize(), DL, MVT::i32);
4571 SDValue Cpy =
4572 DAG.getMemcpy(Chain, DL, DstAddr, Arg, SizeNode,
4573 Outs[i].Flags.getNonZeroByValAlign(),
4574 Outs[i].Flags.getNonZeroByValAlign(),
4575 /*isVol = */ false, /*AlwaysInline = */ true,
4576 /*CI=*/nullptr, std::nullopt, DstInfo,
4578
4579 MemOpChains.push_back(Cpy);
4580 } else {
4581 SDValue Store =
4582 DAG.getStore(Chain, DL, Arg, DstAddr, DstInfo, Alignment);
4583 MemOpChains.push_back(Store);
4584 }
4585 }
4586 }
4587
4588 if (!MemOpChains.empty())
4589 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
4590
4591 SDValue ReadFirstLaneID =
4592 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
4593
4594 SDValue TokenGlue;
4595 if (CLI.ConvergenceControlToken) {
4596 TokenGlue = DAG.getNode(ISD::CONVERGENCECTRL_GLUE, DL, MVT::Glue,
4598 }
4599
4600 // Build a sequence of copy-to-reg nodes chained together with token chain
4601 // and flag operands which copy the outgoing args into the appropriate regs.
4602 SDValue InGlue;
4603
4604 unsigned ArgIdx = 0;
4605 for (auto [Reg, Val] : RegsToPass) {
4606 if (ArgIdx++ >= NumSpecialInputs &&
4607 (IsChainCallConv || !Val->isDivergent()) && TRI->isSGPRPhysReg(Reg)) {
4608 // For chain calls, the inreg arguments are required to be
4609 // uniform. Speculatively Insert a readfirstlane in case we cannot prove
4610 // they are uniform.
4611 //
4612 // For other calls, if an inreg arguments is known to be uniform,
4613 // speculatively insert a readfirstlane in case it is in a VGPR.
4614 //
4615 // FIXME: We need to execute this in a waterfall loop if it is a divergent
4616 // value, so let that continue to produce invalid code.
4617
4618 SmallVector<SDValue, 3> ReadfirstlaneArgs({ReadFirstLaneID, Val});
4619 if (TokenGlue)
4620 ReadfirstlaneArgs.push_back(TokenGlue);
4622 ReadfirstlaneArgs);
4623 }
4624
4625 Chain = DAG.getCopyToReg(Chain, DL, Reg, Val, InGlue);
4626 InGlue = Chain.getValue(1);
4627 }
4628
4629 // We don't usually want to end the call-sequence here because we would tidy
4630 // the frame up *after* the call, however in the ABI-changing tail-call case
4631 // we've carefully laid out the parameters so that when sp is reset they'll be
4632 // in the correct location.
4633 if (IsTailCall && !IsSibCall) {
4634 Chain = DAG.getCALLSEQ_END(Chain, NumBytes, 0, InGlue, DL);
4635 InGlue = Chain.getValue(1);
4636 }
4637
4638 std::vector<SDValue> Ops({Chain});
4639
4640 // Add a redundant copy of the callee global which will not be legalized, as
4641 // we need direct access to the callee later.
4643 const GlobalValue *GV = GSD->getGlobal();
4644 Ops.push_back(Callee);
4645 Ops.push_back(DAG.getTargetGlobalAddress(GV, DL, MVT::i64));
4646 } else {
4647 if (IsTailCall) {
4648 // isEligibleForTailCallOptimization considered whether the call target is
4649 // divergent, but we may still end up with a uniform value in a VGPR.
4650 // Insert a readfirstlane just in case.
4651 SDValue ReadFirstLaneID =
4652 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
4653
4654 SmallVector<SDValue, 3> ReadfirstlaneArgs({ReadFirstLaneID, Callee});
4655 if (TokenGlue)
4656 ReadfirstlaneArgs.push_back(TokenGlue); // Wire up convergence token.
4657 Callee = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, Callee.getValueType(),
4658 ReadfirstlaneArgs);
4659 }
4660
4661 Ops.push_back(Callee);
4662 Ops.push_back(DAG.getTargetConstant(0, DL, MVT::i64));
4663 }
4664
4665 if (IsTailCall) {
4666 // Each tail call may have to adjust the stack by a different amount, so
4667 // this information must travel along with the operation for eventual
4668 // consumption by emitEpilogue.
4669 Ops.push_back(DAG.getTargetConstant(FPDiff, DL, MVT::i32));
4670 }
4671
4672 if (IsChainCallConv)
4673 llvm::append_range(Ops, ChainCallSpecialArgs);
4674
4675 // Add argument registers to the end of the list so that they are known live
4676 // into the call.
4677 for (auto &[Reg, Val] : RegsToPass)
4678 Ops.push_back(DAG.getRegister(Reg, Val.getValueType()));
4679
4680 // Add a register mask operand representing the call-preserved registers.
4681 const uint32_t *Mask = TRI->getCallPreservedMask(MF, CallConv);
4682 assert(Mask && "Missing call preserved mask for calling convention");
4683 Ops.push_back(DAG.getRegisterMask(Mask));
4684
4685 if (SDValue Token = CLI.ConvergenceControlToken) {
4687 GlueOps.push_back(Token);
4688 if (InGlue)
4689 GlueOps.push_back(InGlue);
4690
4691 InGlue = SDValue(DAG.getMachineNode(TargetOpcode::CONVERGENCECTRL_GLUE, DL,
4692 MVT::Glue, GlueOps),
4693 0);
4694 }
4695
4696 if (InGlue)
4697 Ops.push_back(InGlue);
4698
4699 // If we're doing a tall call, use a TC_RETURN here rather than an
4700 // actual call instruction.
4701 if (IsTailCall) {
4702 MFI.setHasTailCall();
4703 unsigned OPC = AMDGPUISD::TC_RETURN;
4704 switch (CallConv) {
4706 OPC = AMDGPUISD::TC_RETURN_GFX;
4707 break;
4710 OPC = UsesDynamicVGPRs ? AMDGPUISD::TC_RETURN_CHAIN_DVGPR
4711 : AMDGPUISD::TC_RETURN_CHAIN;
4712 break;
4713 }
4714
4715 // If the caller is a whole wave function, we need to use a special opcode
4716 // so we can patch up EXEC.
4717 if (Info->isWholeWaveFunction())
4718 OPC = AMDGPUISD::TC_RETURN_GFX_WholeWave;
4719
4720 SDValue Ret = DAG.getNode(OPC, DL, MVT::Other, Ops);
4721 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
4722 return Ret;
4723 }
4724
4725 // Returns a chain and a flag for retval copy to use.
4726 SDValue Call = DAG.getNode(AMDGPUISD::CALL, DL, {MVT::Other, MVT::Glue}, Ops);
4727 DAG.addNoMergeSiteInfo(Call.getNode(), CLI.NoMerge);
4728 Chain = Call.getValue(0);
4729 InGlue = Call.getValue(1);
4730
4731 uint64_t CalleePopBytes = NumBytes;
4732 Chain = DAG.getCALLSEQ_END(Chain, 0, CalleePopBytes, InGlue, DL);
4733 if (!Ins.empty())
4734 InGlue = Chain.getValue(1);
4735
4736 // Handle result values, copying them out of physregs into vregs that we
4737 // return.
4738 return LowerCallResult(Chain, InGlue, CallConv, IsVarArg, Ins, DL, DAG,
4739 InVals, /*IsThisReturn=*/false, SDValue());
4740}
4741
4742// This is similar to the default implementation in ExpandDYNAMIC_STACKALLOC,
4743// except for:
4744// 1. Stack growth direction(default: downwards, AMDGPU: upwards), and
4745// 2. Scale size where, scale = wave-reduction(alloca-size) * wave-size
4747 SelectionDAG &DAG) const {
4748 const MachineFunction &MF = DAG.getMachineFunction();
4750
4751 SDLoc dl(Op);
4752 EVT VT = Op.getValueType();
4753 SDValue Chain = Op.getOperand(0);
4754 Register SPReg = Info->getStackPtrOffsetReg();
4755
4756 // Chain the dynamic stack allocation so that it doesn't modify the stack
4757 // pointer when other instructions are using the stack.
4758 Chain = DAG.getCALLSEQ_START(Chain, 0, 0, dl);
4759
4760 SDValue Size = Op.getOperand(1);
4761 SDValue BaseAddr = DAG.getCopyFromReg(Chain, dl, SPReg, VT);
4762 Align Alignment = cast<ConstantSDNode>(Op.getOperand(2))->getAlignValue();
4763
4764 const TargetFrameLowering *TFL = Subtarget->getFrameLowering();
4766 "Stack grows upwards for AMDGPU");
4767
4768 Chain = BaseAddr.getValue(1);
4769 // When using flat-scratch, the stack offset is unscaled.
4770 const bool HasFlatScratch = Subtarget->hasFlatScratchEnabled();
4771 const unsigned WavefrontSizeLog2 = Subtarget->getWavefrontSizeLog2();
4772
4773 Align StackAlign = TFL->getStackAlign();
4774 if (Alignment > StackAlign) {
4775 uint64_t ScaledAlignment = Alignment.value()
4776 << (HasFlatScratch ? 0 : WavefrontSizeLog2);
4777 uint64_t StackAlignMask = ScaledAlignment - 1;
4778 SDValue TmpAddr = DAG.getNode(ISD::ADD, dl, VT, BaseAddr,
4779 DAG.getConstant(StackAlignMask, dl, VT));
4780 BaseAddr = DAG.getNode(ISD::AND, dl, VT, TmpAddr,
4781 DAG.getSignedConstant(-ScaledAlignment, dl, VT));
4782 }
4783
4784 assert(Size.getValueType() == MVT::i32 && "Size must be 32-bit");
4785 SDValue NewSP;
4787 // Increase the stack pointer by the size of the alloca.
4788 // If not using flat-scratch, we have to scale the size by the wave-size.
4789 SDValue ScaledSize =
4790 HasFlatScratch
4791 ? Size
4792 : DAG.getNode(ISD::SHL, dl, VT, Size,
4793 DAG.getConstant(WavefrontSizeLog2, dl, MVT::i32));
4794 NewSP = DAG.getNode(ISD::ADD, dl, VT, BaseAddr, ScaledSize); // Value
4795 } else {
4796 // For dynamic sized alloca, perform wave-wide reduction to get max of
4797 // alloca size(divergent), and then scale it (when not using flat-scratch)
4798 // by wave-size.
4799 SDValue WaveReduction =
4800 DAG.getTargetConstant(Intrinsic::amdgcn_wave_reduce_umax, dl, MVT::i32);
4801 Size = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::i32, WaveReduction,
4802 Size, DAG.getTargetConstant(0, dl, MVT::i32));
4803 SDValue ScaledSize = Size;
4804 if (!HasFlatScratch) {
4805 ScaledSize =
4806 DAG.getNode(ISD::SHL, dl, VT, Size,
4807 DAG.getConstant(WavefrontSizeLog2, dl, MVT::i32));
4808 }
4809 NewSP =
4810 DAG.getNode(ISD::ADD, dl, VT, BaseAddr, ScaledSize); // Value in vgpr.
4811 SDValue ReadFirstLaneID =
4812 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, dl, MVT::i32);
4813 NewSP = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, dl, MVT::i32, ReadFirstLaneID,
4814 NewSP);
4815 }
4816
4817 Chain = DAG.getCopyToReg(Chain, dl, SPReg, NewSP); // Output chain
4818 SDValue CallSeqEnd = DAG.getCALLSEQ_END(Chain, 0, 0, SDValue(), dl);
4819
4820 return DAG.getMergeValues({BaseAddr, CallSeqEnd}, dl);
4821}
4822
4824 if (Op.getValueType() != MVT::i32)
4825 return Op; // Defer to cannot select error.
4826
4828 SDLoc SL(Op);
4829
4830 SDValue CopyFromSP = DAG.getCopyFromReg(Op->getOperand(0), SL, SP, MVT::i32);
4831
4832 // Convert from wave uniform to swizzled vector address. This should protect
4833 // from any edge cases where the stacksave result isn't directly used with
4834 // stackrestore.
4835 SDValue VectorAddress =
4836 DAG.getNode(AMDGPUISD::WAVE_ADDRESS, SL, MVT::i32, CopyFromSP);
4837 return DAG.getMergeValues({VectorAddress, CopyFromSP.getValue(1)}, SL);
4838}
4839
4841 SelectionDAG &DAG) const {
4842 SDLoc SL(Op);
4843 assert(Op.getValueType() == MVT::i32);
4844
4845 uint32_t BothRoundHwReg =
4847 SDValue GetRoundBothImm = DAG.getTargetConstant(BothRoundHwReg, SL, MVT::i32);
4848
4849 SDValue IntrinID =
4850 DAG.getTargetConstant(Intrinsic::amdgcn_s_getreg, SL, MVT::i32);
4851 SDValue GetReg = DAG.getNode(ISD::INTRINSIC_W_CHAIN, SL, Op->getVTList(),
4852 Op.getOperand(0), IntrinID, GetRoundBothImm);
4853
4854 // There are two rounding modes, one for f32 and one for f64/f16. We only
4855 // report in the standard value range if both are the same.
4856 //
4857 // The raw values also differ from the expected FLT_ROUNDS values. Nearest
4858 // ties away from zero is not supported, and the other values are rotated by
4859 // 1.
4860 //
4861 // If the two rounding modes are not the same, report a target defined value.
4862
4863 // Mode register rounding mode fields:
4864 //
4865 // [1:0] Single-precision round mode.
4866 // [3:2] Double/Half-precision round mode.
4867 //
4868 // 0=nearest even; 1= +infinity; 2= -infinity, 3= toward zero.
4869 //
4870 // Hardware Spec
4871 // Toward-0 3 0
4872 // Nearest Even 0 1
4873 // +Inf 1 2
4874 // -Inf 2 3
4875 // NearestAway0 N/A 4
4876 //
4877 // We have to handle 16 permutations of a 4-bit value, so we create a 64-bit
4878 // table we can index by the raw hardware mode.
4879 //
4880 // (trunc (FltRoundConversionTable >> MODE.fp_round)) & 0xf
4881
4882 SDValue BitTable =
4884
4885 SDValue Two = DAG.getConstant(2, SL, MVT::i32);
4886 SDValue RoundModeTimesNumBits =
4887 DAG.getNode(ISD::SHL, SL, MVT::i32, GetReg, Two);
4888
4889 // TODO: We could possibly avoid a 64-bit shift and use a simpler table if we
4890 // knew only one mode was demanded.
4891 SDValue TableValue =
4892 DAG.getNode(ISD::SRL, SL, MVT::i64, BitTable, RoundModeTimesNumBits);
4893 SDValue TruncTable = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, TableValue);
4894
4895 SDValue EntryMask = DAG.getConstant(0xf, SL, MVT::i32);
4896 SDValue TableEntry =
4897 DAG.getNode(ISD::AND, SL, MVT::i32, TruncTable, EntryMask);
4898
4899 // There's a gap in the 4-bit encoded table and actual enum values, so offset
4900 // if it's an extended value.
4901 SDValue Four = DAG.getConstant(4, SL, MVT::i32);
4902 SDValue IsStandardValue =
4903 DAG.getSetCC(SL, MVT::i1, TableEntry, Four, ISD::SETULT);
4904 SDValue EnumOffset = DAG.getNode(ISD::ADD, SL, MVT::i32, TableEntry, Four);
4905 SDValue Result = DAG.getNode(ISD::SELECT, SL, MVT::i32, IsStandardValue,
4906 TableEntry, EnumOffset);
4907
4908 return DAG.getMergeValues({Result, GetReg.getValue(1)}, SL);
4909}
4910
4912 SelectionDAG &DAG) const {
4913 SDLoc SL(Op);
4914
4915 SDValue NewMode = Op.getOperand(1);
4916 assert(NewMode.getValueType() == MVT::i32);
4917
4918 // Index a table of 4-bit entries mapping from the C FLT_ROUNDS values to the
4919 // hardware MODE.fp_round values.
4920 if (auto *ConstMode = dyn_cast<ConstantSDNode>(NewMode)) {
4921 uint32_t ClampedVal = std::min(
4922 static_cast<uint32_t>(ConstMode->getZExtValue()),
4924 NewMode = DAG.getConstant(
4925 AMDGPU::decodeFltRoundToHWConversionTable(ClampedVal), SL, MVT::i32);
4926 } else {
4927 // If we know the input can only be one of the supported standard modes in
4928 // the range 0-3, we can use a simplified mapping to hardware values.
4929 KnownBits KB = DAG.computeKnownBits(NewMode);
4930 const bool UseReducedTable = KB.countMinLeadingZeros() >= 30;
4931 // The supported standard values are 0-3. The extended values start at 8. We
4932 // need to offset by 4 if the value is in the extended range.
4933
4934 if (UseReducedTable) {
4935 // Truncate to the low 32-bits.
4936 SDValue BitTable = DAG.getConstant(
4937 AMDGPU::FltRoundToHWConversionTable & 0xffff, SL, MVT::i32);
4938
4939 SDValue Two = DAG.getConstant(2, SL, MVT::i32);
4940 SDValue RoundModeTimesNumBits =
4941 DAG.getNode(ISD::SHL, SL, MVT::i32, NewMode, Two);
4942
4943 NewMode =
4944 DAG.getNode(ISD::SRL, SL, MVT::i32, BitTable, RoundModeTimesNumBits);
4945
4946 // TODO: SimplifyDemandedBits on the setreg source here can likely reduce
4947 // the table extracted bits into inline immediates.
4948 } else {
4949 // table_index = umin(value, value - 4)
4950 // MODE.fp_round = (bit_table >> (table_index << 2)) & 0xf
4951 SDValue BitTable =
4953
4954 SDValue Four = DAG.getConstant(4, SL, MVT::i32);
4955 SDValue OffsetEnum = DAG.getNode(ISD::SUB, SL, MVT::i32, NewMode, Four);
4956 SDValue IndexVal =
4957 DAG.getNode(ISD::UMIN, SL, MVT::i32, NewMode, OffsetEnum);
4958
4959 SDValue Two = DAG.getConstant(2, SL, MVT::i32);
4960 SDValue RoundModeTimesNumBits =
4961 DAG.getNode(ISD::SHL, SL, MVT::i32, IndexVal, Two);
4962
4963 SDValue TableValue =
4964 DAG.getNode(ISD::SRL, SL, MVT::i64, BitTable, RoundModeTimesNumBits);
4965 SDValue TruncTable = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, TableValue);
4966
4967 // No need to mask out the high bits since the setreg will ignore them
4968 // anyway.
4969 NewMode = TruncTable;
4970 }
4971
4972 // Insert a readfirstlane in case the value is a VGPR. We could do this
4973 // earlier and keep more operations scalar, but that interferes with
4974 // combining the source.
4975 SDValue ReadFirstLaneID =
4976 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, SL, MVT::i32);
4977 NewMode = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
4978 ReadFirstLaneID, NewMode);
4979 }
4980
4981 // N.B. The setreg will be later folded into s_round_mode on supported
4982 // targets.
4983 SDValue IntrinID =
4984 DAG.getTargetConstant(Intrinsic::amdgcn_s_setreg, SL, MVT::i32);
4985 uint32_t BothRoundHwReg =
4987 SDValue RoundBothImm = DAG.getTargetConstant(BothRoundHwReg, SL, MVT::i32);
4988
4989 SDValue SetReg =
4990 DAG.getNode(ISD::INTRINSIC_VOID, SL, Op->getVTList(), Op.getOperand(0),
4991 IntrinID, RoundBothImm, NewMode);
4992
4993 return SetReg;
4994}
4995
4997 if (Op->isDivergent() &&
4998 (!Subtarget->hasVmemPrefInsts() || !Op.getConstantOperandVal(4)))
4999 // Cannot do I$ prefetch with divergent pointer.
5000 return SDValue();
5001
5002 switch (cast<MemSDNode>(Op)->getAddressSpace()) {
5006 break;
5008 if (Subtarget->hasSafeSmemPrefetch())
5009 break;
5010 [[fallthrough]];
5011 default:
5012 return SDValue();
5013 }
5014
5015 // I$ prefetch
5016 if (!Subtarget->hasSafeSmemPrefetch() && !Op.getConstantOperandVal(4))
5017 return SDValue();
5018
5019 return Op;
5020}
5021
5022// Work around DAG legality rules only based on the result type.
5024 bool IsStrict = Op.getOpcode() == ISD::STRICT_FP_EXTEND;
5025 SDValue Src = Op.getOperand(IsStrict ? 1 : 0);
5026 EVT SrcVT = Src.getValueType();
5027
5028 if (SrcVT.getScalarType() != MVT::bf16)
5029 return Op;
5030
5031 SDLoc SL(Op);
5032 SDValue BitCast =
5033 DAG.getNode(ISD::BITCAST, SL, SrcVT.changeTypeToInteger(), Src);
5034
5035 EVT DstVT = Op.getValueType();
5036 if (IsStrict)
5037 llvm_unreachable("Need STRICT_BF16_TO_FP");
5038
5039 return DAG.getNode(ISD::BF16_TO_FP, SL, DstVT, BitCast);
5040}
5041
5043 SDLoc SL(Op);
5044 if (Op.getValueType() != MVT::i64)
5045 return Op;
5046
5047 uint32_t ModeHwReg =
5049 SDValue ModeHwRegImm = DAG.getTargetConstant(ModeHwReg, SL, MVT::i32);
5050 uint32_t TrapHwReg =
5052 SDValue TrapHwRegImm = DAG.getTargetConstant(TrapHwReg, SL, MVT::i32);
5053
5054 SDVTList VTList = DAG.getVTList(MVT::i32, MVT::Other);
5055 SDValue IntrinID =
5056 DAG.getTargetConstant(Intrinsic::amdgcn_s_getreg, SL, MVT::i32);
5057 SDValue GetModeReg = DAG.getNode(ISD::INTRINSIC_W_CHAIN, SL, VTList,
5058 Op.getOperand(0), IntrinID, ModeHwRegImm);
5059 SDValue GetTrapReg = DAG.getNode(ISD::INTRINSIC_W_CHAIN, SL, VTList,
5060 Op.getOperand(0), IntrinID, TrapHwRegImm);
5061 SDValue TokenReg =
5062 DAG.getNode(ISD::TokenFactor, SL, MVT::Other, GetModeReg.getValue(1),
5063 GetTrapReg.getValue(1));
5064
5065 SDValue CvtPtr =
5066 DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, GetModeReg, GetTrapReg);
5067 SDValue Result = DAG.getNode(ISD::BITCAST, SL, MVT::i64, CvtPtr);
5068
5069 return DAG.getMergeValues({Result, TokenReg}, SL);
5070}
5071
5073 SDLoc SL(Op);
5074 if (Op.getOperand(1).getValueType() != MVT::i64)
5075 return Op;
5076
5077 SDValue Input = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Op.getOperand(1));
5078 SDValue NewModeReg = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Input,
5079 DAG.getConstant(0, SL, MVT::i32));
5080 SDValue NewTrapReg = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Input,
5081 DAG.getConstant(1, SL, MVT::i32));
5082
5083 SDValue ReadFirstLaneID =
5084 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, SL, MVT::i32);
5085 NewModeReg = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
5086 ReadFirstLaneID, NewModeReg);
5087 NewTrapReg = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
5088 ReadFirstLaneID, NewTrapReg);
5089
5090 unsigned ModeHwReg =
5092 SDValue ModeHwRegImm = DAG.getTargetConstant(ModeHwReg, SL, MVT::i32);
5093 unsigned TrapHwReg =
5095 SDValue TrapHwRegImm = DAG.getTargetConstant(TrapHwReg, SL, MVT::i32);
5096
5097 SDValue IntrinID =
5098 DAG.getTargetConstant(Intrinsic::amdgcn_s_setreg, SL, MVT::i32);
5099 SDValue SetModeReg =
5100 DAG.getNode(ISD::INTRINSIC_VOID, SL, MVT::Other, Op.getOperand(0),
5101 IntrinID, ModeHwRegImm, NewModeReg);
5102 SDValue SetTrapReg =
5103 DAG.getNode(ISD::INTRINSIC_VOID, SL, MVT::Other, Op.getOperand(0),
5104 IntrinID, TrapHwRegImm, NewTrapReg);
5105 return DAG.getNode(ISD::TokenFactor, SL, MVT::Other, SetTrapReg, SetModeReg);
5106}
5107
5109 const MachineFunction &MF) const {
5110 Register Reg =
5112 .Case("m0", AMDGPU::M0)
5113 .Case("exec", AMDGPU::EXEC)
5114 .Case("exec_lo", AMDGPU::EXEC_LO)
5115 .Case("exec_hi", AMDGPU::EXEC_HI)
5116 .Case("flat_scratch", AMDGPU::FLAT_SCR)
5117 .Case("flat_scratch_lo", AMDGPU::FLAT_SCR_LO)
5118 .Case("flat_scratch_hi", AMDGPU::FLAT_SCR_HI)
5119 .Case("src_flat_scratch_base", AMDGPU::SRC_FLAT_SCRATCH_BASE)
5120 .Case("src_flat_scratch_base_lo", AMDGPU::SRC_FLAT_SCRATCH_BASE_LO)
5121 .Case("src_flat_scratch_base_hi", AMDGPU::SRC_FLAT_SCRATCH_BASE_HI)
5122 .Default(Register());
5123 if (!Reg)
5124 return Reg;
5125
5126 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
5127 if (!Subtarget->hasFlatScrRegister() &&
5128 TRI->regsOverlap(Reg, AMDGPU::FLAT_SCR))
5129 return Register();
5130
5131 if (!Subtarget->hasGloballyAddressableScratch() &&
5132 TRI->regsOverlap(Reg, AMDGPU::SRC_FLAT_SCRATCH_BASE))
5133 return Register();
5134
5135 switch (Reg) {
5136 case AMDGPU::M0:
5137 case AMDGPU::EXEC_LO:
5138 case AMDGPU::EXEC_HI:
5139 case AMDGPU::FLAT_SCR_LO:
5140 case AMDGPU::FLAT_SCR_HI:
5141 case AMDGPU::SRC_FLAT_SCRATCH_BASE_LO:
5142 case AMDGPU::SRC_FLAT_SCRATCH_BASE_HI:
5143 if (VT.getSizeInBits() == 32)
5144 return Reg;
5145 break;
5146 case AMDGPU::EXEC:
5147 case AMDGPU::FLAT_SCR:
5148 case AMDGPU::SRC_FLAT_SCRATCH_BASE:
5149 if (VT.getSizeInBits() == 64)
5150 return Reg;
5151 break;
5152 default:
5153 llvm_unreachable("missing register type checking");
5154 }
5155
5157 Twine("invalid type for register \"" + StringRef(RegName) + "\"."));
5158}
5159
5160// If kill is not the last instruction, split the block so kill is always a
5161// proper terminator.
5164 MachineBasicBlock *BB) const {
5165 MachineBasicBlock *SplitBB = BB->splitAt(MI, /*UpdateLiveIns=*/true);
5167 MI.setDesc(TII->getKillTerminatorFromPseudo(MI.getOpcode()));
5168 return SplitBB;
5169}
5170
5171// Split block \p MBB at \p MI, as to insert a loop. If \p InstInLoop is true,
5172// \p MI will be the only instruction in the loop body block. Otherwise, it will
5173// be the first instruction in the remainder block.
5174//
5175/// \returns { LoopBody, Remainder }
5176static std::pair<MachineBasicBlock *, MachineBasicBlock *>
5178 MachineFunction *MF = MBB.getParent();
5180
5181 // To insert the loop we need to split the block. Move everything after this
5182 // point to a new block, and insert a new empty block between the two.
5184 MachineBasicBlock *RemainderBB = MF->CreateMachineBasicBlock();
5186 ++MBBI;
5187
5188 MF->insert(MBBI, LoopBB);
5189 MF->insert(MBBI, RemainderBB);
5190
5191 LoopBB->addSuccessor(LoopBB);
5192 LoopBB->addSuccessor(RemainderBB);
5193
5194 // Move the rest of the block into a new block.
5195 RemainderBB->transferSuccessorsAndUpdatePHIs(&MBB);
5196
5197 if (InstInLoop) {
5198 auto Next = std::next(I);
5199
5200 // Move instruction to loop body.
5201 LoopBB->splice(LoopBB->begin(), &MBB, I, Next);
5202
5203 // Move the rest of the block.
5204 RemainderBB->splice(RemainderBB->begin(), &MBB, Next, MBB.end());
5205 } else {
5206 RemainderBB->splice(RemainderBB->begin(), &MBB, I, MBB.end());
5207 }
5208
5209 MBB.addSuccessor(LoopBB);
5210
5211 return std::pair(LoopBB, RemainderBB);
5212}
5213
5214/// Insert \p MI into a BUNDLE with an S_WAITCNT 0 immediately following it.
5216 MachineBasicBlock *MBB = MI.getParent();
5218 auto I = MI.getIterator();
5219 auto E = std::next(I);
5220
5221 // clang-format off
5222 BuildMI(*MBB, E, MI.getDebugLoc(), TII->get(AMDGPU::S_WAITCNT))
5223 .addImm(0);
5224 // clang-format on
5225
5226 MIBundleBuilder Bundler(*MBB, I, E);
5227 finalizeBundle(*MBB, Bundler.begin());
5228}
5229
5232 MachineBasicBlock *BB) const {
5233 const DebugLoc &DL = MI.getDebugLoc();
5234
5236
5238
5239 // Apparently kill flags are only valid if the def is in the same block?
5240 if (MachineOperand *Src = TII->getNamedOperand(MI, AMDGPU::OpName::data0))
5241 Src->setIsKill(false);
5242
5243 auto [LoopBB, RemainderBB] = splitBlockForLoop(MI, *BB, true);
5244
5245 MachineBasicBlock::iterator I = LoopBB->end();
5246
5247 const unsigned EncodedReg = AMDGPU::Hwreg::HwregEncoding::encode(
5249
5250 // Clear TRAP_STS.MEM_VIOL
5251 BuildMI(*LoopBB, LoopBB->begin(), DL, TII->get(AMDGPU::S_SETREG_IMM32_B32))
5252 .addImm(0)
5253 .addImm(EncodedReg);
5254
5256
5257 Register Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
5258
5259 // Load and check TRAP_STS.MEM_VIOL
5260 BuildMI(*LoopBB, I, DL, TII->get(AMDGPU::S_GETREG_B32), Reg)
5261 .addImm(EncodedReg);
5262
5263 // FIXME: Do we need to use an isel pseudo that may clobber scc?
5264 BuildMI(*LoopBB, I, DL, TII->get(AMDGPU::S_CMP_LG_U32))
5265 .addReg(Reg, RegState::Kill)
5266 .addImm(0);
5267 // clang-format off
5268 BuildMI(*LoopBB, I, DL, TII->get(AMDGPU::S_CBRANCH_SCC1))
5269 .addMBB(LoopBB);
5270 // clang-format on
5271
5272 return RemainderBB;
5273}
5274
5275// Do a v_movrels_b32 or v_movreld_b32 for each unique value of \p IdxReg in the
5276// wavefront. If the value is uniform and just happens to be in a VGPR, this
5277// will only do one iteration. In the worst case, this will loop 64 times.
5278//
5279// TODO: Just use v_readlane_b32 if we know the VGPR has a uniform value.
5282 MachineBasicBlock &OrigBB, MachineBasicBlock &LoopBB,
5283 const DebugLoc &DL, const MachineOperand &Idx,
5284 unsigned InitReg, unsigned ResultReg, unsigned PhiReg,
5285 unsigned InitSaveExecReg, int Offset, bool UseGPRIdxMode,
5286 Register &SGPRIdxReg) {
5287
5288 MachineFunction *MF = OrigBB.getParent();
5289 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
5290 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5293
5294 const TargetRegisterClass *BoolRC = TRI->getBoolRC();
5295 Register PhiExec = MRI.createVirtualRegister(BoolRC);
5296 Register NewExec = MRI.createVirtualRegister(BoolRC);
5297 Register CurrentIdxReg =
5298 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
5299 Register CondReg = MRI.createVirtualRegister(BoolRC);
5300
5301 BuildMI(LoopBB, I, DL, TII->get(TargetOpcode::PHI), PhiReg)
5302 .addReg(InitReg)
5303 .addMBB(&OrigBB)
5304 .addReg(ResultReg)
5305 .addMBB(&LoopBB);
5306
5307 BuildMI(LoopBB, I, DL, TII->get(TargetOpcode::PHI), PhiExec)
5308 .addReg(InitSaveExecReg)
5309 .addMBB(&OrigBB)
5310 .addReg(NewExec)
5311 .addMBB(&LoopBB);
5312
5313 // Read the next variant <- also loop target.
5314 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), CurrentIdxReg)
5315 .addReg(Idx.getReg(), getUndefRegState(Idx.isUndef()));
5316
5317 // Compare the just read M0 value to all possible Idx values.
5318 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::V_CMP_EQ_U32_e64), CondReg)
5319 .addReg(CurrentIdxReg)
5320 .addReg(Idx.getReg(), {}, Idx.getSubReg());
5321
5322 // Update EXEC, save the original EXEC value to VCC.
5323 BuildMI(LoopBB, I, DL, TII->get(LMC.AndSaveExecOpc), NewExec)
5324 .addReg(CondReg, RegState::Kill)
5325 .setOperandDead(3); // Dead scc
5326
5327 MRI.setSimpleHint(NewExec, CondReg);
5328
5329 if (UseGPRIdxMode) {
5330 if (Offset == 0) {
5331 SGPRIdxReg = CurrentIdxReg;
5332 } else {
5333 SGPRIdxReg = MRI.createVirtualRegister(&AMDGPU::SGPR_32RegClass);
5334 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_ADD_I32), SGPRIdxReg)
5335 .addReg(CurrentIdxReg, RegState::Kill)
5336 .addImm(Offset)
5337 .setOperandDead(3); // Dead scc
5338 }
5339 } else {
5340 // Move index from VCC into M0
5341 if (Offset == 0) {
5342 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::COPY), AMDGPU::M0)
5343 .addReg(CurrentIdxReg, RegState::Kill);
5344 } else {
5345 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_ADD_I32), AMDGPU::M0)
5346 .addReg(CurrentIdxReg, RegState::Kill)
5347 .addImm(Offset)
5348 .setOperandDead(3); // Dead scc
5349 }
5350 }
5351
5352 // Update EXEC, switch all done bits to 0 and all todo bits to 1.
5353 MachineInstr *InsertPt =
5354 BuildMI(LoopBB, I, DL, TII->get(LMC.XorTermOpc), LMC.ExecReg)
5355 .addReg(LMC.ExecReg)
5356 .addReg(NewExec)
5357 .setOperandDead(3); // Dead scc
5358
5359 // XXX - s_xor_b64 sets scc to 1 if the result is nonzero, so can we use
5360 // s_cbranch_scc0?
5361
5362 // Loop back to V_READFIRSTLANE_B32 if there are still variants to cover.
5363 // clang-format off
5364 BuildMI(LoopBB, I, DL, TII->get(AMDGPU::S_CBRANCH_EXECNZ))
5365 .addMBB(&LoopBB);
5366 // clang-format on
5367
5368 return InsertPt->getIterator();
5369}
5370
5371// This has slightly sub-optimal regalloc when the source vector is killed by
5372// the read. The register allocator does not understand that the kill is
5373// per-workitem, so is kept alive for the whole loop so we end up not re-using a
5374// subregister from it, using 1 more VGPR than necessary. This was saved when
5375// this was expanded after register allocation.
5378 unsigned InitResultReg, unsigned PhiReg, int Offset,
5379 bool UseGPRIdxMode, Register &SGPRIdxReg) {
5380 MachineFunction *MF = MBB.getParent();
5381 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
5382 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5383 MachineRegisterInfo &MRI = MF->getRegInfo();
5384 const DebugLoc &DL = MI.getDebugLoc();
5386
5387 const auto *BoolXExecRC = TRI->getWaveMaskRegClass();
5388 Register DstReg = MI.getOperand(0).getReg();
5389 Register SaveExec = MRI.createVirtualRegister(BoolXExecRC);
5390 Register TmpExec = MRI.createVirtualRegister(BoolXExecRC);
5392
5393 BuildMI(MBB, I, DL, TII->get(TargetOpcode::IMPLICIT_DEF), TmpExec);
5394
5395 // Save the EXEC mask
5396 // clang-format off
5397 BuildMI(MBB, I, DL, TII->get(LMC.MovOpc), SaveExec)
5398 .addReg(LMC.ExecReg);
5399 // clang-format on
5400
5401 auto [LoopBB, RemainderBB] = splitBlockForLoop(MI, MBB, false);
5402
5403 const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
5404
5405 auto InsPt = emitLoadM0FromVGPRLoop(TII, MRI, MBB, *LoopBB, DL, *Idx,
5406 InitResultReg, DstReg, PhiReg, TmpExec,
5407 Offset, UseGPRIdxMode, SGPRIdxReg);
5408
5409 MachineBasicBlock *LandingPad = MF->CreateMachineBasicBlock();
5411 ++MBBI;
5412 MF->insert(MBBI, LandingPad);
5413 LoopBB->removeSuccessor(RemainderBB);
5414 LandingPad->addSuccessor(RemainderBB);
5415 LoopBB->addSuccessor(LandingPad);
5416 MachineBasicBlock::iterator First = LandingPad->begin();
5417 // clang-format off
5418 BuildMI(*LandingPad, First, DL, TII->get(LMC.MovOpc), LMC.ExecReg)
5419 .addReg(SaveExec);
5420 // clang-format on
5421
5422 return InsPt;
5423}
5424
5425// Returns subreg index, offset
5426static std::pair<unsigned, int>
5428 const TargetRegisterClass *SuperRC, unsigned VecReg,
5429 int Offset) {
5430 int NumElts = TRI.getRegSizeInBits(*SuperRC) / 32;
5431
5432 // Skip out of bounds offsets, or else we would end up using an undefined
5433 // register.
5434 if (Offset >= NumElts || Offset < 0)
5435 return std::pair(AMDGPU::sub0, Offset);
5436
5437 return std::pair(SIRegisterInfo::getSubRegFromChannel(Offset), 0);
5438}
5439
5442 int Offset) {
5443 MachineBasicBlock *MBB = MI.getParent();
5444 const DebugLoc &DL = MI.getDebugLoc();
5446
5447 const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
5448
5449 assert(Idx->getReg() != AMDGPU::NoRegister);
5450
5451 if (Offset == 0) {
5452 // clang-format off
5453 BuildMI(*MBB, I, DL, TII->get(AMDGPU::COPY), AMDGPU::M0)
5454 .add(*Idx);
5455 // clang-format on
5456 } else {
5457 BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), AMDGPU::M0)
5458 .add(*Idx)
5459 .addImm(Offset)
5460 .setOperandDead(3); // Dead scc
5461 }
5462}
5463
5466 int Offset) {
5467 MachineBasicBlock *MBB = MI.getParent();
5468 const DebugLoc &DL = MI.getDebugLoc();
5470
5471 const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
5472
5473 if (Offset == 0)
5474 return Idx->getReg();
5475
5476 Register Tmp = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
5477 BuildMI(*MBB, I, DL, TII->get(AMDGPU::S_ADD_I32), Tmp)
5478 .add(*Idx)
5479 .addImm(Offset)
5480 .setOperandDead(3); // Dead scc
5481 return Tmp;
5482}
5483
5486 const GCNSubtarget &ST) {
5487 const SIInstrInfo *TII = ST.getInstrInfo();
5488 const SIRegisterInfo &TRI = TII->getRegisterInfo();
5489 MachineFunction *MF = MBB.getParent();
5490 MachineRegisterInfo &MRI = MF->getRegInfo();
5491
5492 Register Dst = MI.getOperand(0).getReg();
5493 const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
5494 Register SrcReg = TII->getNamedOperand(MI, AMDGPU::OpName::src)->getReg();
5495 int Offset = TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm();
5496
5497 const TargetRegisterClass *VecRC = MRI.getRegClass(SrcReg);
5498 const TargetRegisterClass *IdxRC = MRI.getRegClass(Idx->getReg());
5499
5500 unsigned SubReg;
5501 std::tie(SubReg, Offset) =
5502 computeIndirectRegAndOffset(TRI, VecRC, SrcReg, Offset);
5503
5504 const bool UseGPRIdxMode = ST.useVGPRIndexMode();
5505
5506 // Check for a SGPR index.
5507 if (TII->getRegisterInfo().isSGPRClass(IdxRC)) {
5509 const DebugLoc &DL = MI.getDebugLoc();
5510
5511 if (UseGPRIdxMode) {
5512 // TODO: Look at the uses to avoid the copy. This may require rescheduling
5513 // to avoid interfering with other uses, so probably requires a new
5514 // optimization pass.
5515 Register Idx = getIndirectSGPRIdx(TII, MRI, MI, Offset);
5516
5517 const MCInstrDesc &GPRIDXDesc =
5518 TII->getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), true);
5519 BuildMI(MBB, I, DL, GPRIDXDesc, Dst)
5520 .addReg(SrcReg)
5521 .addReg(Idx)
5522 .addImm(SubReg);
5523 } else {
5525
5526 BuildMI(MBB, I, DL, TII->get(AMDGPU::V_MOVRELS_B32_e32), Dst)
5527 .addReg(SrcReg, {}, SubReg)
5528 .addReg(SrcReg, RegState::Implicit);
5529 }
5530
5531 MI.eraseFromParent();
5532
5533 return &MBB;
5534 }
5535
5536 // Control flow needs to be inserted if indexing with a VGPR.
5537 const DebugLoc &DL = MI.getDebugLoc();
5539
5540 Register PhiReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
5541 Register InitReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
5542
5543 BuildMI(MBB, I, DL, TII->get(TargetOpcode::IMPLICIT_DEF), InitReg);
5544
5545 Register SGPRIdxReg;
5546 auto InsPt = loadM0FromVGPR(TII, MBB, MI, InitReg, PhiReg, Offset,
5547 UseGPRIdxMode, SGPRIdxReg);
5548
5549 MachineBasicBlock *LoopBB = InsPt->getParent();
5550
5551 if (UseGPRIdxMode) {
5552 const MCInstrDesc &GPRIDXDesc =
5553 TII->getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), true);
5554
5555 BuildMI(*LoopBB, InsPt, DL, GPRIDXDesc, Dst)
5556 .addReg(SrcReg)
5557 .addReg(SGPRIdxReg)
5558 .addImm(SubReg);
5559 } else {
5560 BuildMI(*LoopBB, InsPt, DL, TII->get(AMDGPU::V_MOVRELS_B32_e32), Dst)
5561 .addReg(SrcReg, {}, SubReg)
5562 .addReg(SrcReg, RegState::Implicit);
5563 }
5564
5565 MI.eraseFromParent();
5566
5567 return LoopBB;
5568}
5569
5572 const GCNSubtarget &ST) {
5573 const SIInstrInfo *TII = ST.getInstrInfo();
5574 const SIRegisterInfo &TRI = TII->getRegisterInfo();
5575 MachineFunction *MF = MBB.getParent();
5576 MachineRegisterInfo &MRI = MF->getRegInfo();
5577
5578 Register Dst = MI.getOperand(0).getReg();
5579 const MachineOperand *SrcVec = TII->getNamedOperand(MI, AMDGPU::OpName::src);
5580 const MachineOperand *Idx = TII->getNamedOperand(MI, AMDGPU::OpName::idx);
5581 const MachineOperand *Val = TII->getNamedOperand(MI, AMDGPU::OpName::val);
5582 int Offset = TII->getNamedOperand(MI, AMDGPU::OpName::offset)->getImm();
5583 const TargetRegisterClass *VecRC = MRI.getRegClass(SrcVec->getReg());
5584 const TargetRegisterClass *IdxRC = MRI.getRegClass(Idx->getReg());
5585
5586 // This can be an immediate, but will be folded later.
5587 assert(Val->getReg());
5588
5589 unsigned SubReg;
5590 std::tie(SubReg, Offset) =
5591 computeIndirectRegAndOffset(TRI, VecRC, SrcVec->getReg(), Offset);
5592 const bool UseGPRIdxMode = ST.useVGPRIndexMode();
5593
5594 if (Idx->getReg() == AMDGPU::NoRegister) {
5596 const DebugLoc &DL = MI.getDebugLoc();
5597
5598 assert(Offset == 0);
5599
5600 BuildMI(MBB, I, DL, TII->get(TargetOpcode::INSERT_SUBREG), Dst)
5601 .add(*SrcVec)
5602 .add(*Val)
5603 .addImm(SubReg);
5604
5605 MI.eraseFromParent();
5606 return &MBB;
5607 }
5608
5609 // Check for a SGPR index.
5610 if (TII->getRegisterInfo().isSGPRClass(IdxRC)) {
5612 const DebugLoc &DL = MI.getDebugLoc();
5613
5614 if (UseGPRIdxMode) {
5615 Register Idx = getIndirectSGPRIdx(TII, MRI, MI, Offset);
5616
5617 const MCInstrDesc &GPRIDXDesc =
5618 TII->getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), false);
5619 BuildMI(MBB, I, DL, GPRIDXDesc, Dst)
5620 .addReg(SrcVec->getReg())
5621 .add(*Val)
5622 .addReg(Idx)
5623 .addImm(SubReg);
5624 } else {
5626
5627 const MCInstrDesc &MovRelDesc = TII->getIndirectRegWriteMovRelPseudo(
5628 TRI.getRegSizeInBits(*VecRC), 32, false);
5629 BuildMI(MBB, I, DL, MovRelDesc, Dst)
5630 .addReg(SrcVec->getReg())
5631 .add(*Val)
5632 .addImm(SubReg);
5633 }
5634 MI.eraseFromParent();
5635 return &MBB;
5636 }
5637
5638 // Control flow needs to be inserted if indexing with a VGPR.
5639 if (Val->isReg())
5640 MRI.clearKillFlags(Val->getReg());
5641
5642 const DebugLoc &DL = MI.getDebugLoc();
5643
5644 Register PhiReg = MRI.createVirtualRegister(VecRC);
5645
5646 Register SGPRIdxReg;
5647 auto InsPt = loadM0FromVGPR(TII, MBB, MI, SrcVec->getReg(), PhiReg, Offset,
5648 UseGPRIdxMode, SGPRIdxReg);
5649 MachineBasicBlock *LoopBB = InsPt->getParent();
5650
5651 if (UseGPRIdxMode) {
5652 const MCInstrDesc &GPRIDXDesc =
5653 TII->getIndirectGPRIDXPseudo(TRI.getRegSizeInBits(*VecRC), false);
5654
5655 BuildMI(*LoopBB, InsPt, DL, GPRIDXDesc, Dst)
5656 .addReg(PhiReg)
5657 .add(*Val)
5658 .addReg(SGPRIdxReg)
5659 .addImm(SubReg);
5660 } else {
5661 const MCInstrDesc &MovRelDesc = TII->getIndirectRegWriteMovRelPseudo(
5662 TRI.getRegSizeInBits(*VecRC), 32, false);
5663 BuildMI(*LoopBB, InsPt, DL, MovRelDesc, Dst)
5664 .addReg(PhiReg)
5665 .add(*Val)
5666 .addImm(SubReg);
5667 }
5668
5669 MI.eraseFromParent();
5670 return LoopBB;
5671}
5672
5674 MachineBasicBlock *BB) {
5675 // For targets older than GFX12, we emit a sequence of 32-bit operations.
5676 // For GFX12, we emit s_add_u64 and s_sub_u64.
5677 MachineFunction *MF = BB->getParent();
5678 const SIInstrInfo *TII = MF->getSubtarget<GCNSubtarget>().getInstrInfo();
5679 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
5681 const DebugLoc &DL = MI.getDebugLoc();
5682 MachineOperand &Dest = MI.getOperand(0);
5683 MachineOperand &Src0 = MI.getOperand(1);
5684 MachineOperand &Src1 = MI.getOperand(2);
5685 bool IsAdd = (MI.getOpcode() == AMDGPU::S_ADD_U64_PSEUDO);
5686 if (ST.hasScalarAddSub64()) {
5687 // FIXME: If scc is used, this deletes the def
5688 unsigned Opc = IsAdd ? AMDGPU::S_ADD_U64 : AMDGPU::S_SUB_U64;
5689 // clang-format off
5690 BuildMI(*BB, MI, DL, TII->get(Opc), Dest.getReg())
5691 .add(Src0)
5692 .add(Src1);
5693 // clang-format on
5694 } else {
5695 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5696 const TargetRegisterClass *BoolRC = TRI->getBoolRC();
5697
5698 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
5699 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
5700
5701 MachineOperand Src0Sub0 = TII->buildExtractSubRegOrImm(
5702 MI, MRI, Src0, BoolRC, AMDGPU::sub0, &AMDGPU::SReg_32RegClass);
5703 MachineOperand Src0Sub1 = TII->buildExtractSubRegOrImm(
5704 MI, MRI, Src0, BoolRC, AMDGPU::sub1, &AMDGPU::SReg_32RegClass);
5705
5706 MachineOperand Src1Sub0 = TII->buildExtractSubRegOrImm(
5707 MI, MRI, Src1, BoolRC, AMDGPU::sub0, &AMDGPU::SReg_32RegClass);
5708 MachineOperand Src1Sub1 = TII->buildExtractSubRegOrImm(
5709 MI, MRI, Src1, BoolRC, AMDGPU::sub1, &AMDGPU::SReg_32RegClass);
5710
5711 const MachineOperand &ImpDefSCC = MI.getOperand(3);
5712 assert(ImpDefSCC.getReg() == AMDGPU::SCC && ImpDefSCC.isDef());
5713
5714 unsigned LoOpc = IsAdd ? AMDGPU::S_ADD_U32 : AMDGPU::S_SUB_U32;
5715 unsigned HiOpc = IsAdd ? AMDGPU::S_ADDC_U32 : AMDGPU::S_SUBB_U32;
5716 BuildMI(*BB, MI, DL, TII->get(LoOpc), DestSub0).add(Src0Sub0).add(Src1Sub0);
5717 auto Hi = BuildMI(*BB, MI, DL, TII->get(HiOpc), DestSub1)
5718 .add(Src0Sub1)
5719 .add(Src1Sub1);
5720 if (ImpDefSCC.isDead())
5721 Hi.setOperandDead(3);
5722 BuildMI(*BB, MI, DL, TII->get(TargetOpcode::REG_SEQUENCE), Dest.getReg())
5723 .addReg(DestSub0)
5724 .addImm(AMDGPU::sub0)
5725 .addReg(DestSub1)
5726 .addImm(AMDGPU::sub1);
5727 }
5728 MI.eraseFromParent();
5729 return BB;
5730}
5731
5733 MachineFunction *MF = BB->getParent();
5734 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
5735 const SIInstrInfo *TII = ST.getInstrInfo();
5736 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5737 MachineRegisterInfo &MRI = MF->getRegInfo();
5738 const DebugLoc &DL = MI.getDebugLoc();
5739 Register Dst = MI.getOperand(0).getReg();
5740 const MachineOperand &Src0 = MI.getOperand(1);
5741 const MachineOperand &Src1 = MI.getOperand(2);
5742 Register SrcCond = MI.getOperand(3).getReg();
5743
5744 Register DstLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
5745 Register DstHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
5746 const TargetRegisterClass *CondRC = TRI->getWaveMaskRegClass();
5747 Register SrcCondCopy = MRI.createVirtualRegister(CondRC);
5748
5749 int Src0Idx =
5750 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src0);
5751 int Src1Idx =
5752 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src1);
5753 const TargetRegisterClass *Src0RC =
5754 TRI->getAllocatableClass(TII->getRegClass(MI.getDesc(), Src0Idx));
5755 const TargetRegisterClass *Src1RC =
5756 TRI->getAllocatableClass(TII->getRegClass(MI.getDesc(), Src1Idx));
5757
5758 const TargetRegisterClass *Src0SubRC =
5759 TRI->getSubRegisterClass(Src0RC, AMDGPU::sub0);
5760 const TargetRegisterClass *Src1SubRC =
5761 TRI->getSubRegisterClass(Src1RC, AMDGPU::sub1);
5762
5763 MachineOperand Src0Sub0 = TII->buildExtractSubRegOrImm(
5764 MI, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
5765 MachineOperand Src1Sub0 = TII->buildExtractSubRegOrImm(
5766 MI, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
5767
5768 MachineOperand Src0Sub1 = TII->buildExtractSubRegOrImm(
5769 MI, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
5770 MachineOperand Src1Sub1 = TII->buildExtractSubRegOrImm(
5771 MI, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
5772
5773 BuildMI(*BB, MI, DL, TII->get(AMDGPU::COPY), SrcCondCopy).addReg(SrcCond);
5774 BuildMI(*BB, MI, DL, TII->get(AMDGPU::V_CNDMASK_B32_e64), DstLo)
5775 .addImm(0)
5776 .add(Src0Sub0)
5777 .addImm(0)
5778 .add(Src1Sub0)
5779 .addReg(SrcCondCopy);
5780
5781 BuildMI(*BB, MI, DL, TII->get(AMDGPU::V_CNDMASK_B32_e64), DstHi)
5782 .addImm(0)
5783 .add(Src0Sub1)
5784 .addImm(0)
5785 .add(Src1Sub1)
5786 .addReg(SrcCondCopy);
5787
5788 BuildMI(*BB, MI, DL, TII->get(AMDGPU::REG_SEQUENCE), Dst)
5789 .addReg(DstLo)
5790 .addImm(AMDGPU::sub0)
5791 .addReg(DstHi)
5792 .addImm(AMDGPU::sub1);
5793 MI.eraseFromParent();
5794}
5795
5797 switch (Opc) {
5798 case AMDGPU::S_MIN_U32:
5799 return std::numeric_limits<uint32_t>::max();
5800 case AMDGPU::S_MIN_I32:
5801 return std::numeric_limits<int32_t>::max();
5802 case AMDGPU::S_MAX_U32:
5803 return std::numeric_limits<uint32_t>::min();
5804 case AMDGPU::S_MAX_I32:
5805 return std::numeric_limits<int32_t>::min();
5806 case AMDGPU::V_ADD_F32_e64: // -0.0
5807 return 0x80000000;
5808 case AMDGPU::V_SUB_F32_e64: // +0.0
5809 return 0x0;
5810 case AMDGPU::S_ADD_I32:
5811 case AMDGPU::S_SUB_I32:
5812 case AMDGPU::S_OR_B32:
5813 case AMDGPU::S_XOR_B32:
5814 return std::numeric_limits<uint32_t>::min();
5815 case AMDGPU::S_AND_B32:
5816 return std::numeric_limits<uint32_t>::max();
5817 case AMDGPU::V_MIN_F32_e64:
5818 case AMDGPU::V_MAX_F32_e64:
5819 return 0x7fc00000; // qNAN
5820 case AMDGPU::V_CMP_LT_U64_e64: // umin.u64
5821 return std::numeric_limits<uint64_t>::max();
5822 case AMDGPU::V_CMP_LT_I64_e64: // min.i64
5823 return std::numeric_limits<int64_t>::max();
5824 case AMDGPU::V_CMP_GT_U64_e64: // umax.u64
5825 return std::numeric_limits<uint64_t>::min();
5826 case AMDGPU::V_CMP_GT_I64_e64: // max.i64
5827 return std::numeric_limits<int64_t>::min();
5828 case AMDGPU::V_MIN_F64_e64:
5829 case AMDGPU::V_MAX_F64_e64:
5830 case AMDGPU::V_MIN_NUM_F64_e64:
5831 case AMDGPU::V_MAX_NUM_F64_e64:
5832 return 0x7FF8000000000000; // qNAN
5833 case AMDGPU::S_ADD_U64_PSEUDO:
5834 case AMDGPU::S_SUB_U64_PSEUDO:
5835 case AMDGPU::S_OR_B64:
5836 case AMDGPU::S_XOR_B64:
5837 return std::numeric_limits<uint64_t>::min();
5838 case AMDGPU::S_AND_B64:
5839 return std::numeric_limits<uint64_t>::max();
5840 case AMDGPU::V_ADD_F64_e64:
5841 case AMDGPU::V_ADD_F64_pseudo_e64:
5842 return 0x8000000000000000; // -0.0
5843 default:
5844 llvm_unreachable("Unexpected opcode in getIdentityValueForWaveReduction");
5845 }
5846}
5847
5848static bool is32bitWaveReduceOperation(unsigned Opc) {
5849 return Opc == AMDGPU::S_MIN_U32 || Opc == AMDGPU::S_MIN_I32 ||
5850 Opc == AMDGPU::S_MAX_U32 || Opc == AMDGPU::S_MAX_I32 ||
5851 Opc == AMDGPU::S_ADD_I32 || Opc == AMDGPU::S_SUB_I32 ||
5852 Opc == AMDGPU::S_AND_B32 || Opc == AMDGPU::S_OR_B32 ||
5853 Opc == AMDGPU::S_XOR_B32 || Opc == AMDGPU::V_MIN_F32_e64 ||
5854 Opc == AMDGPU::V_MAX_F32_e64 || Opc == AMDGPU::V_ADD_F32_e64 ||
5855 Opc == AMDGPU::V_SUB_F32_e64;
5856}
5857
5859 return Opc == AMDGPU::V_MIN_F32_e64 || Opc == AMDGPU::V_MAX_F32_e64 ||
5860 Opc == AMDGPU::V_ADD_F32_e64 || Opc == AMDGPU::V_SUB_F32_e64 ||
5861 Opc == AMDGPU::V_MIN_F64_e64 || Opc == AMDGPU::V_MAX_F64_e64 ||
5862 Opc == AMDGPU::V_MIN_NUM_F64_e64 || Opc == AMDGPU::V_MAX_NUM_F64_e64 ||
5863 Opc == AMDGPU::V_ADD_F64_e64 || Opc == AMDGPU::V_ADD_F64_pseudo_e64;
5864}
5865
5866static std::tuple<unsigned, unsigned>
5868 unsigned DPPOpc;
5869 switch (Opc) {
5870 case AMDGPU::S_MIN_U32:
5871 DPPOpc = AMDGPU::V_MIN_U32_dpp;
5872 break;
5873 case AMDGPU::S_MIN_I32:
5874 DPPOpc = AMDGPU::V_MIN_I32_dpp;
5875 break;
5876 case AMDGPU::S_MAX_U32:
5877 DPPOpc = AMDGPU::V_MAX_U32_dpp;
5878 break;
5879 case AMDGPU::S_MAX_I32:
5880 DPPOpc = AMDGPU::V_MAX_I32_dpp;
5881 break;
5882 case AMDGPU::S_ADD_I32:
5883 case AMDGPU::S_SUB_I32:
5884 DPPOpc = ST.hasAddNoCarryInsts() ? AMDGPU::V_ADD_U32_dpp
5885 : AMDGPU::V_ADD_CO_U32_dpp;
5886 break;
5887 case AMDGPU::S_AND_B32:
5888 DPPOpc = AMDGPU::V_AND_B32_dpp;
5889 break;
5890 case AMDGPU::S_OR_B32:
5891 DPPOpc = AMDGPU::V_OR_B32_dpp;
5892 break;
5893 case AMDGPU::S_XOR_B32:
5894 DPPOpc = AMDGPU::V_XOR_B32_dpp;
5895 break;
5896 case AMDGPU::V_ADD_F32_e64:
5897 case AMDGPU::V_SUB_F32_e64:
5898 DPPOpc = AMDGPU::V_ADD_F32_dpp;
5899 break;
5900 case AMDGPU::V_MIN_F32_e64:
5901 DPPOpc = AMDGPU::V_MIN_F32_dpp;
5902 break;
5903 case AMDGPU::V_MAX_F32_e64:
5904 DPPOpc = AMDGPU::V_MAX_F32_dpp;
5905 break;
5906 case AMDGPU::V_CMP_LT_U64_e64: // umin.u64
5907 case AMDGPU::V_CMP_LT_I64_e64: // min.i64
5908 case AMDGPU::V_CMP_GT_U64_e64: // umax.u64
5909 case AMDGPU::V_CMP_GT_I64_e64: // max.i64
5910 case AMDGPU::S_ADD_U64_PSEUDO:
5911 case AMDGPU::S_SUB_U64_PSEUDO:
5912 case AMDGPU::S_AND_B64:
5913 case AMDGPU::S_OR_B64:
5914 case AMDGPU::S_XOR_B64:
5915 case AMDGPU::V_MIN_NUM_F64_e64:
5916 case AMDGPU::V_MIN_F64_e64:
5917 case AMDGPU::V_MAX_NUM_F64_e64:
5918 case AMDGPU::V_MAX_F64_e64:
5919 case AMDGPU::V_ADD_F64_pseudo_e64:
5920 case AMDGPU::V_ADD_F64_e64:
5921 DPPOpc = AMDGPU::V_MOV_B64_DPP_PSEUDO;
5922 break;
5923 default:
5924 llvm_unreachable("unhandled lane op");
5925 }
5926 unsigned ClampOpc = Opc;
5927 if (!ST.getInstrInfo()->isVALU(Opc, /*AllowLDSDMA=*/true)) {
5928 if (Opc == AMDGPU::S_SUB_I32)
5929 ClampOpc = AMDGPU::S_ADD_I32;
5930 if (Opc == AMDGPU::S_ADD_U64_PSEUDO || Opc == AMDGPU::S_SUB_U64_PSEUDO)
5931 ClampOpc = AMDGPU::V_ADD_CO_U32_e64;
5932 else if (Opc == AMDGPU::S_AND_B64)
5933 ClampOpc = AMDGPU::V_AND_B32_e64;
5934 else if (Opc == AMDGPU::S_OR_B64)
5935 ClampOpc = AMDGPU::V_OR_B32_e64;
5936 else if (Opc == AMDGPU::S_XOR_B64)
5937 ClampOpc = AMDGPU::V_XOR_B32_e64;
5938 else
5939 ClampOpc = ST.getInstrInfo()->getVALUOp(ClampOpc);
5940 }
5941 return {DPPOpc, ClampOpc};
5942}
5943
5944static std::pair<Register, Register>
5946 const TargetRegisterClass *SrcRC, const GCNSubtarget &ST,
5947 MachineRegisterInfo &MRI) {
5948 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5949 const SIInstrInfo *TII = ST.getInstrInfo();
5950 const TargetRegisterClass *SrcSubRC =
5951 TRI->getSubRegisterClass(SrcRC, AMDGPU::sub0);
5952 Register Op1L =
5953 TII->buildExtractSubReg(MI, MRI, Op, SrcRC, AMDGPU::sub0, SrcSubRC);
5954 Register Op1H =
5955 TII->buildExtractSubReg(MI, MRI, Op, SrcRC, AMDGPU::sub1, SrcSubRC);
5956 return {Op1L, Op1H};
5957}
5958
5961 const GCNSubtarget &ST,
5962 unsigned Opc) {
5964 const SIRegisterInfo *TRI = ST.getRegisterInfo();
5965 const DebugLoc &DL = MI.getDebugLoc();
5966 const SIInstrInfo *TII = ST.getInstrInfo();
5967
5968 // Reduction operations depend on whether the input operand is SGPR or VGPR.
5969 Register SrcReg = MI.getOperand(1).getReg();
5970 bool isSGPR = TRI->isSGPRClass(MRI.getRegClass(SrcReg));
5971 Register DstReg = MI.getOperand(0).getReg();
5972 unsigned Stratergy = static_cast<unsigned>(MI.getOperand(2).getImm());
5973 enum WAVE_REDUCE_STRATEGY : unsigned { DEFAULT = 0, ITERATIVE = 1, DPP = 2 };
5974 MachineBasicBlock *RetBB = nullptr;
5975 unsigned MIOpc = MI.getOpcode();
5976 auto BuildRegSequence = [&](MachineBasicBlock &BB,
5978 Register Src0, Register Src1) {
5979 auto RegSequence =
5980 BuildMI(BB, MI, DL, TII->get(TargetOpcode::REG_SEQUENCE), Dst)
5981 .addReg(Src0)
5982 .addImm(AMDGPU::sub0)
5983 .addReg(Src1)
5984 .addImm(AMDGPU::sub1);
5985 return RegSequence;
5986 };
5987 if (isSGPR) {
5988 switch (Opc) {
5989 case AMDGPU::S_MIN_U32:
5990 case AMDGPU::S_MIN_I32:
5991 case AMDGPU::V_MIN_F32_e64:
5992 case AMDGPU::S_MAX_U32:
5993 case AMDGPU::S_MAX_I32:
5994 case AMDGPU::V_MAX_F32_e64:
5995 case AMDGPU::S_AND_B32:
5996 case AMDGPU::S_OR_B32: {
5997 // Idempotent operations.
5998 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MOV_B32), DstReg).addReg(SrcReg);
5999 RetBB = &BB;
6000 break;
6001 }
6002 case AMDGPU::V_CMP_LT_U64_e64: // umin
6003 case AMDGPU::V_CMP_LT_I64_e64: // min
6004 case AMDGPU::V_CMP_GT_U64_e64: // umax
6005 case AMDGPU::V_CMP_GT_I64_e64: // max
6006 case AMDGPU::V_MIN_F64_e64:
6007 case AMDGPU::V_MIN_NUM_F64_e64:
6008 case AMDGPU::V_MAX_F64_e64:
6009 case AMDGPU::V_MAX_NUM_F64_e64:
6010 case AMDGPU::S_AND_B64:
6011 case AMDGPU::S_OR_B64: {
6012 // Idempotent operations.
6013 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MOV_B64), DstReg).addReg(SrcReg);
6014 RetBB = &BB;
6015 break;
6016 }
6017 case AMDGPU::S_XOR_B32:
6018 case AMDGPU::S_XOR_B64:
6019 case AMDGPU::S_ADD_I32:
6020 case AMDGPU::S_ADD_U64_PSEUDO:
6021 case AMDGPU::V_ADD_F32_e64:
6022 case AMDGPU::V_ADD_F64_e64:
6023 case AMDGPU::V_ADD_F64_pseudo_e64:
6024 case AMDGPU::S_SUB_I32:
6025 case AMDGPU::S_SUB_U64_PSEUDO:
6026 case AMDGPU::V_SUB_F32_e64: {
6027 const TargetRegisterClass *WaveMaskRegClass = TRI->getWaveMaskRegClass();
6028 const TargetRegisterClass *DstRegClass = MRI.getRegClass(DstReg);
6029 Register ExecMask = MRI.createVirtualRegister(WaveMaskRegClass);
6030 Register NumActiveLanes =
6031 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6032
6033 bool IsWave32 = ST.isWave32();
6034 unsigned MovOpc = IsWave32 ? AMDGPU::S_MOV_B32 : AMDGPU::S_MOV_B64;
6035 MCRegister ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
6036 unsigned BitCountOpc =
6037 IsWave32 ? AMDGPU::S_BCNT1_I32_B32 : AMDGPU::S_BCNT1_I32_B64;
6038
6039 BuildMI(BB, MI, DL, TII->get(MovOpc), ExecMask).addReg(ExecReg);
6040
6041 auto NewAccumulator =
6042 BuildMI(BB, MI, DL, TII->get(BitCountOpc), NumActiveLanes)
6043 .addReg(ExecMask)
6044 .setOperandDead(2); // Dead scc
6045
6046 switch (Opc) {
6047 case AMDGPU::S_XOR_B32:
6048 case AMDGPU::S_XOR_B64: {
6049 // Performing an XOR operation on a uniform value
6050 // depends on the parity of the number of active lanes.
6051 // For even parity, the result will be 0, for odd
6052 // parity the result will be the same as the input value.
6053 Register ParityRegister =
6054 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6055 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_AND_B32), ParityRegister)
6056 .addReg(NewAccumulator->getOperand(0).getReg())
6057 .addImm(1)
6058 .setOperandDead(3); // Dead scc
6059 // Check if Src is a known identity constant.
6060 MachineInstr *SrcDef = MRI.getVRegDef(SrcReg);
6061 if (SrcDef && SrcDef->isMoveImmediate()) {
6062 int64_t Imm = SrcDef->getOperand(1).getImm();
6063 if (Imm == 1) { // 1 * parity(exec) = parity(exec)
6064 if (Opc == AMDGPU::S_XOR_B32) {
6065 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MOV_B32), DstReg)
6066 .addReg(ParityRegister);
6067 } else {
6068 Register DstHi =
6069 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6070 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MOV_B32), DstHi).addImm(0);
6071 BuildRegSequence(BB, MI, DstReg, ParityRegister, DstHi);
6072 }
6073 break;
6074 }
6075 }
6076 if (Opc == AMDGPU::S_XOR_B32) {
6077 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DstReg)
6078 .addReg(SrcReg)
6079 .addReg(ParityRegister);
6080 } else {
6081 Register DestSub0 =
6082 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6083 Register DestSub1 =
6084 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6085 auto [Op1L, Op1H] = ExtractSubRegs(MI, MI.getOperand(1),
6086 MRI.getRegClass(SrcReg), ST, MRI);
6087 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DestSub0)
6088 .addReg(Op1L)
6089 .addReg(ParityRegister);
6090 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DestSub1)
6091 .addReg(Op1H)
6092 .addReg(ParityRegister);
6093 BuildRegSequence(BB, MI, DstReg, DestSub0, DestSub1);
6094 }
6095 break;
6096 }
6097 case AMDGPU::S_SUB_I32: {
6098 Register NegatedVal = MRI.createVirtualRegister(DstRegClass);
6099 // Take the negation of the source operand.
6100 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedVal)
6101 .addImm(0)
6102 .addReg(SrcReg)
6103 .setOperandDead(3); // Dead scc
6104 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DstReg)
6105 .addReg(NegatedVal)
6106 .addReg(NewAccumulator->getOperand(0).getReg());
6107 break;
6108 }
6109 case AMDGPU::S_ADD_I32: {
6110 // Check if Src is a known identity constant.
6111 MachineInstr *SrcDef = MRI.getVRegDef(SrcReg);
6112 if (SrcDef && SrcDef->isMoveImmediate()) {
6113 int64_t Imm = SrcDef->getOperand(1).getImm();
6114 if (Imm == 1) { // 1 * bitcount(exec) = bitcount(exec)
6115 BuildMI(BB, MI, DL, TII->get(AMDGPU::COPY), DstReg)
6116 .addReg(NewAccumulator->getOperand(0).getReg());
6117 break;
6118 }
6119 }
6120 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DstReg)
6121 .addReg(SrcReg)
6122 .addReg(NewAccumulator->getOperand(0).getReg());
6123 break;
6124 }
6125 case AMDGPU::S_ADD_U64_PSEUDO:
6126 case AMDGPU::S_SUB_U64_PSEUDO: {
6127 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6128 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6129 Register Op1H_Op0L_Reg =
6130 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6131 Register Op1L_Op0H_Reg =
6132 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6133 Register CarryReg =
6134 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6135 Register AddReg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6136 Register NegatedValLo =
6137 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6138 Register NegatedValHi =
6139 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6140 auto [Op1L, Op1H] = ExtractSubRegs(MI, MI.getOperand(1),
6141 MRI.getRegClass(SrcReg), ST, MRI);
6142 // Check if Src is a known identity constant.
6143 MachineInstr *SrcDef = MRI.getVRegDef(SrcReg);
6144 if (SrcDef && SrcDef->isMoveImmediate()) {
6145 int64_t Imm = SrcDef->getOperand(1).getImm();
6146 if (Imm == 1 && Opc == AMDGPU::S_ADD_U64_PSEUDO) {
6147 // 1 * bitcount(exec) = bitcount(exec)
6148 Register DstHi =
6149 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6150 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MOV_B32), DstHi).addImm(0);
6151 BuildRegSequence(BB, MI, DstReg,
6152 NewAccumulator->getOperand(0).getReg(), DstHi);
6153 break;
6154 }
6155 }
6156 if (Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6157 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedValLo)
6158 .addImm(0)
6159 .addReg(NewAccumulator->getOperand(0).getReg())
6160 .setOperandDead(3); // Dead scc
6161 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_ASHR_I32), NegatedValHi)
6162 .addReg(NegatedValLo)
6163 .addImm(31)
6164 .setOperandDead(3); // Dead scc
6165 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), Op1L_Op0H_Reg)
6166 .addReg(Op1L)
6167 .addReg(NegatedValHi);
6168 }
6169 Register LowOpcode = Opc == AMDGPU::S_SUB_U64_PSEUDO
6170 ? NegatedValLo
6171 : NewAccumulator->getOperand(0).getReg();
6172 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), DestSub0)
6173 .addReg(Op1L)
6174 .addReg(LowOpcode);
6175 if (ST.hasScalarMulHiInsts()) {
6176 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_HI_U32), CarryReg)
6177 .addReg(Op1L)
6178 .addReg(LowOpcode);
6179 } else {
6180 Register VCarryReg =
6181 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6182 Register LowOpVGPR =
6183 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6184 BuildMI(BB, MI, DL, TII->get(AMDGPU::COPY), LowOpVGPR)
6185 .addReg(LowOpcode);
6186 BuildMI(BB, MI, DL, TII->get(AMDGPU::V_MUL_HI_U32_e64), VCarryReg)
6187 .addReg(Op1L)
6188 .addReg(LowOpVGPR);
6189 BuildMI(BB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), CarryReg)
6190 .addReg(VCarryReg);
6191 }
6192 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_MUL_I32), Op1H_Op0L_Reg)
6193 .addReg(Op1H)
6194 .addReg(LowOpcode);
6195
6196 Register HiVal = Opc == AMDGPU::S_SUB_U64_PSEUDO ? AddReg : DestSub1;
6197 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_ADD_U32), HiVal)
6198 .addReg(CarryReg)
6199 .addReg(Op1H_Op0L_Reg)
6200 .setOperandDead(3); // Dead scc
6201
6202 if (Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6203 BuildMI(BB, MI, DL, TII->get(AMDGPU::S_ADD_U32), DestSub1)
6204 .addReg(HiVal)
6205 .addReg(Op1L_Op0H_Reg)
6206 .setOperandDead(3); // Dead scc
6207 }
6208 BuildRegSequence(BB, MI, DstReg, DestSub0, DestSub1);
6209 break;
6210 }
6211 case AMDGPU::V_ADD_F32_e64:
6212 case AMDGPU::V_ADD_F64_e64:
6213 case AMDGPU::V_ADD_F64_pseudo_e64:
6214 case AMDGPU::V_SUB_F32_e64: {
6215 bool is32BitOpc = is32bitWaveReduceOperation(Opc);
6216 const TargetRegisterClass *VregRC = TII->getRegClass(TII->get(Opc), 0);
6217 Register ActiveLanesVreg = MRI.createVirtualRegister(VregRC);
6218 Register DstVreg = MRI.createVirtualRegister(VregRC);
6219 // Get number of active lanes as a float val.
6220 BuildMI(BB, MI, DL,
6221 TII->get(is32BitOpc ? AMDGPU::V_CVT_F32_I32_e64
6222 : AMDGPU::V_CVT_F64_I32_e64),
6223 ActiveLanesVreg)
6224 .addReg(NewAccumulator->getOperand(0).getReg())
6225 .addImm(0) // clamp
6226 .addImm(0); // output-modifier
6227
6228 // Take negation of input for SUB reduction
6229 unsigned srcMod = (MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32 ||
6230 MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64)
6233 unsigned MulOpc = is32BitOpc ? AMDGPU::V_MUL_F32_e64
6234 : ST.getGeneration() >= AMDGPUSubtarget::GFX12
6235 ? AMDGPU::V_MUL_F64_pseudo_e64
6236 : AMDGPU::V_MUL_F64_e64;
6237 auto DestVregInst = BuildMI(BB, MI, DL, TII->get(MulOpc),
6238 DstVreg)
6239 .addImm(srcMod) // src0 modifier
6240 .addReg(SrcReg)
6241 .addImm(SISrcMods::NONE) // src1 modifier
6242 .addReg(ActiveLanesVreg)
6243 .addImm(SISrcMods::NONE) // clamp
6244 .addImm(SISrcMods::NONE); // output-mod
6245 if (is32BitOpc) {
6246 BuildMI(BB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), DstReg)
6247 .addReg(DstVreg);
6248 } else {
6249 Register LaneValueLoReg =
6250 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6251 Register LaneValueHiReg =
6252 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6253 auto [Op1L, Op1H] =
6254 ExtractSubRegs(MI, DestVregInst->getOperand(0), VregRC, ST, MRI);
6255 // lane value input should be in an sgpr
6256 BuildMI(BB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
6257 LaneValueLoReg)
6258 .addReg(Op1L);
6259 BuildMI(BB, MI, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
6260 LaneValueHiReg)
6261 .addReg(Op1H);
6262 NewAccumulator =
6263 BuildRegSequence(BB, MI, DstReg, LaneValueLoReg, LaneValueHiReg);
6264 }
6265 }
6266 }
6267 RetBB = &BB;
6268 }
6269 }
6270 } else {
6272 Register SrcReg = MI.getOperand(1).getReg();
6273 bool is32BitOpc = is32bitWaveReduceOperation(Opc);
6275 bool NeedsMovDPP = !is32BitOpc;
6276 // Create virtual registers required for lowering.
6277 const TargetRegisterClass *WaveMaskRegClass = TRI->getWaveMaskRegClass();
6278 const TargetRegisterClass *DstRegClass = MRI.getRegClass(DstReg);
6279 const TargetRegisterClass *SrcRegClass = MRI.getRegClass(SrcReg);
6280 bool IsWave32 = ST.isWave32();
6281 unsigned MovOpcForExec = IsWave32 ? AMDGPU::S_MOV_B32 : AMDGPU::S_MOV_B64;
6282 unsigned ExecReg = IsWave32 ? AMDGPU::EXEC_LO : AMDGPU::EXEC;
6283 if (Stratergy == WAVE_REDUCE_STRATEGY::ITERATIVE ||
6284 !ST.hasDPP()) { // If target doesn't support DPP operations, default to
6285 // iterative stratergy
6286
6287 // To reduce the VGPR using iterative approach, we need to iterate
6288 // over all the active lanes. Lowering consists of ComputeLoop,
6289 // which iterate over only active lanes. We use copy of EXEC register
6290 // as induction variable and every active lane modifies it using bitset0
6291 // so that we will get the next active lane for next iteration.
6292
6293 // Create Control flow for loop
6294 // Split MI's Machine Basic block into For loop
6295 auto [ComputeLoop, ComputeEnd] = splitBlockForLoop(MI, BB, true);
6296
6297 Register LoopIterator = MRI.createVirtualRegister(WaveMaskRegClass);
6298 Register IdentityValReg = MRI.createVirtualRegister(DstRegClass);
6299 Register AccumulatorReg = MRI.createVirtualRegister(DstRegClass);
6300 Register ActiveBitsReg = MRI.createVirtualRegister(WaveMaskRegClass);
6301 Register NewActiveBitsReg = MRI.createVirtualRegister(WaveMaskRegClass);
6302 Register FF1Reg = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6303 Register LaneValueReg = MRI.createVirtualRegister(DstRegClass);
6304
6305 // Create initial values of induction variable from Exec, Accumulator and
6306 // insert branch instr to newly created ComputeBlock
6307 BuildMI(BB, I, DL, TII->get(MovOpcForExec), LoopIterator).addReg(ExecReg);
6308 uint64_t IdentityValue =
6309 MI.getOpcode() == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64
6310 ? 0x0 // +0.0 for double sub reduction
6312 BuildMI(BB, I, DL,
6313 TII->get(is32BitOpc ? AMDGPU::S_MOV_B32
6314 : AMDGPU::S_MOV_B64_IMM_PSEUDO),
6315 IdentityValReg)
6316 .addImm(IdentityValue);
6317 // clang-format off
6318 BuildMI(BB, I, DL, TII->get(AMDGPU::S_BRANCH))
6319 .addMBB(ComputeLoop);
6320 // clang-format on
6321
6322 // Start constructing ComputeLoop
6323 I = ComputeLoop->begin();
6324 auto Accumulator =
6325 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::PHI), AccumulatorReg)
6326 .addReg(IdentityValReg)
6327 .addMBB(&BB);
6328 auto ActiveBits =
6329 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::PHI), ActiveBitsReg)
6330 .addReg(LoopIterator)
6331 .addMBB(&BB);
6332
6333 I = ComputeLoop->end();
6334 MachineInstr *NewAccumulator;
6335 // Perform the computations
6336 unsigned SFFOpc =
6337 IsWave32 ? AMDGPU::S_FF1_I32_B32 : AMDGPU::S_FF1_I32_B64;
6338 BuildMI(*ComputeLoop, I, DL, TII->get(SFFOpc), FF1Reg)
6339 .addReg(ActiveBitsReg);
6340 if (is32BitOpc) {
6341 Register OpDstReg = DstReg;
6342 bool hasSrc0Modifier = AMDGPU::getNamedOperandIdx(
6343 Opc, AMDGPU::OpName::src0_modifiers) != -1;
6344 bool hasSrc1Modifier = AMDGPU::getNamedOperandIdx(
6345 Opc, AMDGPU::OpName::src1_modifiers) != -1;
6346 bool hasClamp =
6347 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::clamp) != -1;
6348 bool hasOpSel =
6349 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::op_sel) != -1;
6350 bool hasOMod =
6351 AMDGPU::getNamedOperandIdx(Opc, AMDGPU::OpName::omod) != -1;
6352 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READLANE_B32),
6353 LaneValueReg)
6354 .addReg(SrcReg)
6355 .addReg(FF1Reg);
6356 if (ST.getInstrInfo()->isVALU(Opc, /*AllowLDSDMA=*/true)) {
6357 // Get the Lane Value in VGPR to avoid the Constant Bus Restriction
6358 Register LaneValVgpr = MRI.createVirtualRegister(SrcRegClass);
6359 Register VgprResultReg = MRI.createVirtualRegister(SrcRegClass);
6360 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::COPY), LaneValVgpr)
6361 .addReg(LaneValueReg);
6362 OpDstReg = VgprResultReg;
6363 LaneValueReg = LaneValVgpr;
6364 }
6365 auto OpInstr = BuildMI(*ComputeLoop, I, DL, TII->get(Opc), OpDstReg);
6366 if (hasSrc0Modifier)
6367 OpInstr.addImm(SISrcMods::NONE); // src0 modifier
6368 OpInstr.addReg(AccumulatorReg); // src0
6369 if (hasSrc1Modifier)
6370 OpInstr.addImm(SISrcMods::NONE); // src1 modifier
6371 OpInstr.addReg(LaneValueReg); // src1
6372 if (hasClamp)
6373 OpInstr.addImm(0); // clamp
6374 if (hasOpSel)
6375 OpInstr.addImm(0); // opsel
6376 if (hasOMod)
6377 OpInstr.addImm(0); // omod
6378 if (TII->isSALU(Opc))
6379 OpInstr.setOperandDead(3); // Dead scc
6380 if (ST.getInstrInfo()->isVALU(Opc, /*AllowLDSDMA=*/true)) {
6381 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32),
6382 DstReg)
6383 .addReg(OpDstReg);
6384 }
6385 } else {
6386 Register LaneValueLoReg =
6387 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6388 Register LaneValueHiReg =
6389 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6390 Register LaneValReg =
6391 MRI.createVirtualRegister(&AMDGPU::SReg_64RegClass);
6392 auto [Op1L, Op1H] = ExtractSubRegs(MI, MI.getOperand(1),
6393 MRI.getRegClass(SrcReg), ST, MRI);
6394 // lane value input should be in an sgpr
6395 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READLANE_B32),
6396 LaneValueLoReg)
6397 .addReg(Op1L)
6398 .addReg(FF1Reg);
6399 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::V_READLANE_B32),
6400 LaneValueHiReg)
6401 .addReg(Op1H)
6402 .addReg(FF1Reg);
6403 auto LaneValue = BuildRegSequence(*ComputeLoop, I, LaneValReg,
6404 LaneValueLoReg, LaneValueHiReg);
6405 switch (Opc) {
6406 case AMDGPU::S_OR_B64:
6407 case AMDGPU::S_AND_B64:
6408 case AMDGPU::S_XOR_B64: {
6409 NewAccumulator = BuildMI(*ComputeLoop, I, DL, TII->get(Opc), DstReg)
6410 .addReg(Accumulator->getOperand(0).getReg())
6411 .addReg(LaneValue->getOperand(0).getReg())
6412 .setOperandDead(3); // Dead scc
6413 break;
6414 }
6415 case AMDGPU::V_CMP_GT_I64_e64:
6416 case AMDGPU::V_CMP_GT_U64_e64:
6417 case AMDGPU::V_CMP_LT_I64_e64:
6418 case AMDGPU::V_CMP_LT_U64_e64: {
6419 Register LaneMaskReg = MRI.createVirtualRegister(WaveMaskRegClass);
6420 Register ComparisonResultReg =
6421 MRI.createVirtualRegister(WaveMaskRegClass);
6422 int SrcIdx =
6423 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src);
6424 const TargetRegisterClass *VregClass =
6425 TRI->getAllocatableClass(TII->getRegClass(MI.getDesc(), SrcIdx));
6426 Register AccumulatorVReg = MRI.createVirtualRegister(VregClass);
6427 auto [SrcReg0Sub0, SrcReg0Sub1] = ExtractSubRegs(
6428 MI, Accumulator->getOperand(0), VregClass, ST, MRI);
6429 BuildRegSequence(*ComputeLoop, I, AccumulatorVReg, SrcReg0Sub0,
6430 SrcReg0Sub1);
6431 BuildMI(*ComputeLoop, I, DL, TII->get(Opc), LaneMaskReg)
6432 .addReg(LaneValue->getOperand(0).getReg())
6433 .addReg(AccumulatorVReg);
6434
6435 unsigned AndOpc = IsWave32 ? AMDGPU::S_AND_B32 : AMDGPU::S_AND_B64;
6436 BuildMI(*ComputeLoop, I, DL, TII->get(AndOpc), ComparisonResultReg)
6437 .addReg(LaneMaskReg)
6438 .addReg(ActiveBitsReg);
6439
6440 NewAccumulator = BuildMI(*ComputeLoop, I, DL,
6441 TII->get(AMDGPU::S_CSELECT_B64), DstReg)
6442 .addReg(LaneValue->getOperand(0).getReg())
6443 .addReg(Accumulator->getOperand(0).getReg());
6444 break;
6445 }
6446 case AMDGPU::V_MIN_F64_e64:
6447 case AMDGPU::V_MIN_NUM_F64_e64:
6448 case AMDGPU::V_MAX_F64_e64:
6449 case AMDGPU::V_MAX_NUM_F64_e64:
6450 case AMDGPU::V_ADD_F64_e64:
6451 case AMDGPU::V_ADD_F64_pseudo_e64: {
6452 int SrcIdx =
6453 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::src);
6454 const TargetRegisterClass *VregRC =
6455 TRI->getAllocatableClass(TII->getRegClass(MI.getDesc(), SrcIdx));
6456 Register AccumulatorVReg = MRI.createVirtualRegister(VregRC);
6457 Register DstVreg = MRI.createVirtualRegister(VregRC);
6458 Register LaneValLo =
6459 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6460 Register LaneValHi =
6461 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6462 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::COPY), AccumulatorVReg)
6463 .addReg(Accumulator->getOperand(0).getReg());
6464 unsigned Modifier =
6465 MI.getOpcode() == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64
6468 auto DstVregInst =
6469 BuildMI(*ComputeLoop, I, DL, TII->get(Opc), DstVreg)
6470 .addImm(Modifier) // src0 modifiers
6471 .addReg(LaneValue->getOperand(0).getReg())
6472 .addImm(SISrcMods::NONE) // src1 modifiers
6473 .addReg(AccumulatorVReg)
6474 .addImm(SISrcMods::NONE) // clamp
6475 .addImm(SISrcMods::NONE); // omod
6476 auto ReadLaneLo =
6477 BuildMI(*ComputeLoop, I, DL,
6478 TII->get(AMDGPU::V_READFIRSTLANE_B32), LaneValLo);
6479 auto ReadLaneHi =
6480 BuildMI(*ComputeLoop, I, DL,
6481 TII->get(AMDGPU::V_READFIRSTLANE_B32), LaneValHi);
6482 MachineBasicBlock::iterator Iters = *ReadLaneLo;
6483 auto [Op1L, Op1H] = ExtractSubRegs(*Iters, DstVregInst->getOperand(0),
6484 VregRC, ST, MRI);
6485 ReadLaneLo.addReg(Op1L);
6486 ReadLaneHi.addReg(Op1H);
6487 NewAccumulator =
6488 BuildRegSequence(*ComputeLoop, I, DstReg, LaneValLo, LaneValHi);
6489 break;
6490 }
6491 case AMDGPU::S_ADD_U64_PSEUDO:
6492 case AMDGPU::S_SUB_U64_PSEUDO: {
6493 NewAccumulator = BuildMI(*ComputeLoop, I, DL, TII->get(Opc), DstReg)
6494 .addReg(Accumulator->getOperand(0).getReg())
6495 .addReg(LaneValue->getOperand(0).getReg())
6496 .setOperandDead(3); // Dead scc
6497 ComputeLoop =
6498 expand64BitScalarArithmetic(*NewAccumulator, ComputeLoop);
6499 break;
6500 }
6501 }
6502 }
6503 // Manipulate the iterator to get the next active lane
6504 unsigned BITSETOpc =
6505 IsWave32 ? AMDGPU::S_BITSET0_B32 : AMDGPU::S_BITSET0_B64;
6506 BuildMI(*ComputeLoop, I, DL, TII->get(BITSETOpc), NewActiveBitsReg)
6507 .addReg(FF1Reg)
6508 .addReg(ActiveBitsReg);
6509
6510 // Add phi nodes
6511 Accumulator.addReg(DstReg).addMBB(ComputeLoop);
6512 ActiveBits.addReg(NewActiveBitsReg).addMBB(ComputeLoop);
6513
6514 // Creating branching
6515 MachineInstrBuilder SetSCCInstr;
6516 if (!ST.hasScalarCompareEq64()) {
6517 // For targets <= gfx7, use an S_OR_B32/B64 instruction to set SCC.
6518 Register LaneMaskReg = MRI.createVirtualRegister(WaveMaskRegClass);
6519 unsigned CMPOpc = IsWave32 ? AMDGPU::S_OR_B32 : AMDGPU::S_OR_B64;
6520 SetSCCInstr =
6521 BuildMI(*ComputeLoop, I, DL, TII->get(CMPOpc), LaneMaskReg);
6522 } else {
6523 unsigned CMPOpc =
6524 IsWave32 ? AMDGPU::S_CMP_LG_U32 : AMDGPU::S_CMP_LG_U64;
6525 SetSCCInstr = BuildMI(*ComputeLoop, I, DL, TII->get(CMPOpc));
6526 }
6527 SetSCCInstr.addReg(NewActiveBitsReg);
6528 if (ST.hasScalarCompareEq64())
6529 SetSCCInstr.addImm(0);
6530 else
6531 SetSCCInstr.addReg(NewActiveBitsReg);
6532 BuildMI(*ComputeLoop, I, DL, TII->get(AMDGPU::S_CBRANCH_SCC1))
6533 .addMBB(ComputeLoop);
6534
6535 RetBB = ComputeEnd;
6536 } else {
6537 assert(ST.hasDPP() && "Sub Target does not support DPP Operations");
6538 MachineBasicBlock *CurrBB = &BB;
6539 Register SrcWithIdentity = MRI.createVirtualRegister(SrcRegClass);
6540 Register IdentityVGPR = MRI.createVirtualRegister(SrcRegClass);
6541 Register IdentitySGPR = MRI.createVirtualRegister(DstRegClass);
6542 Register DPPRowShr1 = MRI.createVirtualRegister(SrcRegClass);
6543 Register DPPRowShr2 = MRI.createVirtualRegister(SrcRegClass);
6544 Register DPPRowShr4 = MRI.createVirtualRegister(SrcRegClass);
6545 Register DPPRowShr8 = MRI.createVirtualRegister(SrcRegClass);
6546 Register RowBcast15 = MRI.createVirtualRegister(SrcRegClass);
6547 Register ReducedValSGPR = MRI.createVirtualRegister(DstRegClass);
6548 Register NegatedReducedVal = MRI.createVirtualRegister(DstRegClass);
6549 Register RowBcast31 = MRI.createVirtualRegister(SrcRegClass);
6550 Register UndefExec = MRI.createVirtualRegister(WaveMaskRegClass);
6551 Register FinalDPPResult;
6552 MachineInstr *SrcWithIdentityInstr;
6553 MachineInstr *LastBcastInstr;
6554 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::IMPLICIT_DEF), UndefExec);
6555
6557 BuildMI(*CurrBB, MI, DL,
6558 TII->get(is32BitOpc ? AMDGPU::S_MOV_B32
6559 : AMDGPU::S_MOV_B64_IMM_PSEUDO),
6560 IdentitySGPR)
6561 .addImm(IdentityValue);
6562 auto IdentityCopyInstr =
6563 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::COPY), IdentityVGPR)
6564 .addReg(IdentitySGPR);
6565 auto DPPClampOpcPair = getDPPOpcForWaveReduction(Opc, ST);
6566 unsigned DPPOpc = std::get<0>(DPPClampOpcPair);
6567 unsigned ClampOpc = std::get<1>(DPPClampOpcPair);
6568 auto BuildSetInactiveInstr = [&](Register Dst, Register Src0,
6569 Register Src1) {
6570 return BuildMI(BB, MI, DL, TII->get(AMDGPU::V_SET_INACTIVE_B32),
6571 Dst)
6572 .addImm(0) // src0 modifiers
6573 .addReg(Src0) // src0
6574 .addImm(0) // src1 modifiers
6575 .addReg(Src1) // identity value for inactive lanes
6576 .addReg(UndefExec); // bool i1
6577 };
6578 auto BuildDPPMachineInstr = [&](Register Dst, Register Src,
6579 unsigned DPPCtrl) {
6580 auto DPPInstr =
6581 BuildMI(*CurrBB, MI, DL, TII->get(DPPOpc), Dst).addReg(Src); // old
6582 if (isFPOp && !NeedsMovDPP)
6583 DPPInstr.addImm(SISrcMods::NONE); // src0 modifier
6584 DPPInstr.addReg(Src); // src0
6585 if (isFPOp && !NeedsMovDPP)
6586 DPPInstr.addImm(SISrcMods::NONE); // src1 modifier
6587 if (!NeedsMovDPP)
6588 DPPInstr.addReg(Src); // src1
6589 if (AMDGPU::getNamedOperandIdx(DPPOpc, AMDGPU::OpName::clamp) >= 0)
6590 DPPInstr.addImm(0); // clamp
6591 DPPInstr
6592 .addImm(DPPCtrl) // dpp-ctrl
6593 .addImm(0xf) // row-mask
6594 .addImm(0xf) // bank-mask
6595 .addImm(0); // bound-control
6596 };
6597 auto BuildClampInstr = [&](Register Dst, Register Src0, Register Src1,
6598 bool isAddSub = false,
6599 bool needsCarryIn = false,
6600 Register CarryIn = Register()) {
6601 unsigned InstrOpc = ClampOpc;
6602 Register CarryOutReg = MRI.createVirtualRegister(WaveMaskRegClass);
6603 if (needsCarryIn)
6604 InstrOpc = AMDGPU::V_ADDC_U32_e64;
6605 auto ClampInstr = BuildMI(*CurrBB, MI, DL, TII->get(InstrOpc), Dst);
6606 if (isFPOp)
6607 ClampInstr.addImm(SISrcMods::NONE); // src0 mod
6608 if (isAddSub) {
6609 if (needsCarryIn)
6610 ClampInstr.addReg(CarryOutReg,
6612 RegState::Dead); // killed carry-out reg
6613 else
6614 ClampInstr.addReg(CarryOutReg, RegState::Define); // carry-out reg
6615 }
6616 ClampInstr.addReg(Src0); // src0
6617 if (isFPOp)
6618 ClampInstr.addImm(SISrcMods::NONE); // src1 mod
6619 ClampInstr.addReg(Src1); // src1
6620 if (needsCarryIn)
6621 ClampInstr.addReg(CarryIn, RegState::Kill); // carry-in reg
6622 if (AMDGPU::getNamedOperandIdx(InstrOpc, AMDGPU::OpName::clamp) >= 0)
6623 ClampInstr.addImm(0); // clamp
6624 if (isFPOp)
6625 ClampInstr.addImm(0); // omod
6626 LastBcastInstr = ClampInstr;
6627 return CarryOutReg;
6628 };
6629 auto BuildPostDPPInstr = [&](Register Src0, Register Src1) {
6630 bool isAddSubOpc =
6631 Opc == AMDGPU::S_ADD_U64_PSEUDO || Opc == AMDGPU::S_SUB_U64_PSEUDO;
6632 bool isBitWiseOpc = Opc == AMDGPU::S_AND_B64 ||
6633 Opc == AMDGPU::S_OR_B64 || Opc == AMDGPU::S_XOR_B64;
6634 Register ReturnReg = MRI.createVirtualRegister(SrcRegClass);
6635 if (isAddSubOpc || isBitWiseOpc) {
6636 Register ResLo = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6637 Register ResHi = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6638 MachineOperand Src0Operand =
6639 MachineOperand::CreateReg(Src0, /*isDef=*/false);
6640 MachineOperand Src1Operand =
6641 MachineOperand::CreateReg(Src1, /*isDef=*/false);
6642 auto [Src0Lo, Src0Hi] =
6643 ExtractSubRegs(MI, Src0Operand, SrcRegClass, ST, MRI);
6644 auto [Src1Lo, Src1Hi] =
6645 ExtractSubRegs(MI, Src1Operand, SrcRegClass, ST, MRI);
6646 Register CarryReg = BuildClampInstr(
6647 ResLo, Src0Lo, Src1Lo, isAddSubOpc, /*needsCarryIn*/ false);
6648 BuildClampInstr(ResHi, Src0Hi, Src1Hi, isAddSubOpc,
6649 /*needsCarryIn*/ isAddSubOpc, CarryReg);
6650 BuildRegSequence(*CurrBB, MI, ReturnReg, ResLo, ResHi);
6651 } else {
6652 if (isFPOp) {
6653 BuildMI(*CurrBB, MI, DL, TII->get(Opc), ReturnReg)
6654 .addImm(SISrcMods::NONE) // src0 modifiers
6655 .addReg(Src0)
6656 .addImm(SISrcMods::NONE) // src1 modifiers
6657 .addReg(Src1)
6658 .addImm(SISrcMods::NONE) // clamp
6659 .addImm(SISrcMods::NONE); // omod
6660 } else {
6661 Register CmpMaskReg = MRI.createVirtualRegister(WaveMaskRegClass);
6662 BuildMI(*CurrBB, MI, DL, TII->get(Opc), CmpMaskReg)
6663 .addReg(Src0) // src0
6664 .addReg(Src1); // src1
6665 LastBcastInstr =
6666 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_CNDMASK_B64_PSEUDO),
6667 ReturnReg)
6668 .addReg(Src1) // src0
6669 .addReg(Src0) // src1
6670 .addReg(CmpMaskReg); // src2
6671 expand64BitV_CNDMASK(*LastBcastInstr, CurrBB);
6672 }
6673 }
6674 return ReturnReg;
6675 };
6676
6677 // Set inactive lanes to the identity value.
6678 if (is32BitOpc) {
6679 SrcWithIdentityInstr =
6680 BuildSetInactiveInstr(SrcWithIdentity, SrcReg, IdentityVGPR);
6681 } else {
6682 Register SrcWithIdentitylo =
6683 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6684 Register SrcWithIdentityhi =
6685 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6686 auto [Reg0Sub0, Reg0Sub1] = ExtractSubRegs(
6687 MI, IdentityCopyInstr->getOperand(0), SrcRegClass, ST, MRI);
6688 auto [SrcReg0Sub0, SrcReg0Sub1] =
6689 ExtractSubRegs(MI, MI.getOperand(1), SrcRegClass, ST, MRI);
6690 MachineInstr *SetInactiveLoInstr =
6691 BuildSetInactiveInstr(SrcWithIdentitylo, SrcReg0Sub0, Reg0Sub0);
6692 MachineInstr *SetInactiveHiInstr =
6693 BuildSetInactiveInstr(SrcWithIdentityhi, SrcReg0Sub1, Reg0Sub1);
6694 SrcWithIdentityInstr =
6695 BuildRegSequence(*CurrBB, MI, SrcWithIdentity,
6696 SetInactiveLoInstr->getOperand(0).getReg(),
6697 SetInactiveHiInstr->getOperand(0).getReg());
6698 }
6699 // DPP reduction
6700 Register SrcWithIdentityReg =
6701 SrcWithIdentityInstr->getOperand(0).getReg();
6702 BuildDPPMachineInstr(DPPRowShr1, SrcWithIdentityReg,
6704 if (NeedsMovDPP)
6705 DPPRowShr1 = BuildPostDPPInstr(SrcWithIdentityReg, DPPRowShr1);
6706
6707 BuildDPPMachineInstr(DPPRowShr2, DPPRowShr1,
6709 if (NeedsMovDPP)
6710 DPPRowShr2 = BuildPostDPPInstr(DPPRowShr1, DPPRowShr2);
6711
6712 BuildDPPMachineInstr(DPPRowShr4, DPPRowShr2,
6714 if (NeedsMovDPP)
6715 DPPRowShr4 = BuildPostDPPInstr(DPPRowShr2, DPPRowShr4);
6716
6717 BuildDPPMachineInstr(DPPRowShr8, DPPRowShr4,
6719 if (NeedsMovDPP)
6720 DPPRowShr8 = BuildPostDPPInstr(DPPRowShr4, DPPRowShr8);
6721
6722 if (ST.hasDPPBroadcasts()) {
6723 BuildDPPMachineInstr(RowBcast15, DPPRowShr8, AMDGPU::DPP::BCAST15);
6724 if (NeedsMovDPP)
6725 RowBcast15 = BuildPostDPPInstr(DPPRowShr8, RowBcast15);
6726 } else {
6727 // magic constant: 0x1E0
6728 // To Set BIT_MODE : bit 15 = 0
6729 // XOR mask : bit [14:10] = 0
6730 // OR mask : bit [9:5] = 15
6731 // AND mask : bit [4:0] = 0
6732 if (is32BitOpc) {
6733 Register SwizzledValue =
6734 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6735 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_SWIZZLE_B32),
6736 SwizzledValue)
6737 .addReg(DPPRowShr8) // addr
6738 .addImm(0x1E0) // swizzle offset (i16)
6739 .addImm(0x0); // gds (i1)
6740 BuildClampInstr(RowBcast15, DPPRowShr8, SwizzledValue);
6741 } else {
6742 Register SwizzledValuelo =
6743 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6744 Register SwizzledValuehi =
6745 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6746 Register SwizzledValue64 = MRI.createVirtualRegister(SrcRegClass);
6747 MachineOperand DPPRowShr8Op =
6748 MachineOperand::CreateReg(DPPRowShr8, /*isDef=*/false);
6749 auto [Op1L, Op1H] =
6750 ExtractSubRegs(MI, DPPRowShr8Op, SrcRegClass, ST, MRI);
6751 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_SWIZZLE_B32),
6752 SwizzledValuelo)
6753 .addReg(Op1L) // addr
6754 .addImm(0x1E0) // swizzle offset (i16)
6755 .addImm(0x0); // gds (i1)
6756 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_SWIZZLE_B32),
6757 SwizzledValuehi)
6758 .addReg(Op1H) // addr
6759 .addImm(0x1E0) // swizzle offset (i16)
6760 .addImm(0x0); // gds (i1)
6761 BuildRegSequence(*CurrBB, MI, SwizzledValue64, SwizzledValuelo,
6762 SwizzledValuehi);
6763 if (NeedsMovDPP)
6764 RowBcast15 = BuildPostDPPInstr(DPPRowShr8, SwizzledValue64);
6765 else
6766 BuildClampInstr(RowBcast15, DPPRowShr8, SwizzledValue64);
6767 }
6768 }
6769 FinalDPPResult = RowBcast15;
6770 if (!IsWave32) {
6771 if (ST.hasDPPBroadcasts()) {
6772 BuildDPPMachineInstr(RowBcast31, RowBcast15, AMDGPU::DPP::BCAST31);
6773 if (NeedsMovDPP)
6774 RowBcast31 = BuildPostDPPInstr(RowBcast15, RowBcast31);
6775 } else {
6776 Register ShiftedThreadID =
6777 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6778 Register PermuteByteOffset =
6779 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6780 Register PermutedValue = MRI.createVirtualRegister(SrcRegClass);
6781 Register Lane32Offset =
6782 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6783 Register WordSizeConst =
6784 MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
6785 Register ThreadIDRegLo =
6786 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6787 Register ThreadIDReg =
6788 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6789 // Get the thread ID.
6790 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_MBCNT_LO_U32_B32_e64),
6791 ThreadIDRegLo)
6792 .addImm(-1)
6793 .addImm(0);
6794 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_MBCNT_HI_U32_B32_e64),
6795 ThreadIDReg)
6796 .addImm(-1)
6797 .addReg(ThreadIDRegLo);
6798 // shift each lane over by 32 positions, so value in 31st lane is
6799 // present in 63rd lane.
6800 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::S_MOV_B32), Lane32Offset)
6801 .addImm(0x20);
6802 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_ADD_U32_e64),
6803 ShiftedThreadID)
6804 .addReg(ThreadIDReg)
6805 .addReg(Lane32Offset)
6806 .addImm(0); // clamp
6807 // multiply by reg size.
6808 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::S_MOV_B32), WordSizeConst)
6809 .addImm(0x4);
6810 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_MUL_LO_U32_e64),
6811 PermuteByteOffset)
6812 .addReg(WordSizeConst)
6813 .addReg(ShiftedThreadID);
6814 // Permute the lanes
6815 if (is32BitOpc) {
6816 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_PERMUTE_B32),
6817 PermutedValue)
6818 .addReg(PermuteByteOffset) // addr
6819 .addReg(RowBcast15) // data
6820 .addImm(0); // offset
6821 } else {
6822 Register PermutedValuelo =
6823 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6824 Register PermutedValuehi =
6825 MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
6826 MachineOperand RowBcast15Op =
6827 MachineOperand::CreateReg(RowBcast15, /*isDef=*/false);
6828 auto [RowBcast15Lo, RowBcast15Hi] =
6829 ExtractSubRegs(MI, RowBcast15Op, SrcRegClass, ST, MRI);
6830 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_PERMUTE_B32),
6831 PermutedValuelo)
6832 .addReg(PermuteByteOffset) // addr
6833 .addReg(RowBcast15Lo) // swizzle offset (i16)
6834 .addImm(0x0); // gds (i1)
6835 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::DS_PERMUTE_B32),
6836 PermutedValuehi)
6837 .addReg(PermuteByteOffset) // addr
6838 .addReg(RowBcast15Hi) // swizzle offset (i16)
6839 .addImm(0x0); // gds (i1)
6840 BuildRegSequence(*CurrBB, MI, PermutedValue, PermutedValuelo,
6841 PermutedValuehi);
6842 }
6843 if (NeedsMovDPP)
6844 RowBcast31 = BuildPostDPPInstr(RowBcast15, PermutedValue);
6845 else
6846 BuildClampInstr(RowBcast31, RowBcast15, PermutedValue);
6847 }
6848 FinalDPPResult = RowBcast31;
6849 }
6850 if (MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32 ||
6851 MIOpc == AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64) {
6852 Register NegatedValVGPR = MRI.createVirtualRegister(SrcRegClass);
6853 // Opc for f32 reduction is V_SUB_F32.
6854 // For f64, there is no equivalent V_SUB_F64 opcode, so use
6855 // V_ADD_F64/V_ADD_F64_pseudo, and negate the second operand.
6856 BuildMI(*CurrBB, MI, DL, TII->get(Opc),
6857 NegatedValVGPR)
6858 .addImm(SISrcMods::NONE) // src0 mods
6859 .addReg(IdentityVGPR) // src0
6860 .addImm(is32BitOpc ? SISrcMods::NONE : SISrcMods::NEG) // src1 mods
6861 .addReg(IsWave32 ? RowBcast15 : RowBcast31) // src1
6862 .addImm(SISrcMods::NONE) // clamp
6863 .addImm(SISrcMods::NONE); // omod
6864 FinalDPPResult = NegatedValVGPR;
6865 }
6866 // The final reduced value is in the last lane.
6867 if (is32BitOpc) {
6868 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_READLANE_B32),
6869 ReducedValSGPR)
6870 .addReg(FinalDPPResult)
6871 .addImm(ST.getWavefrontSize() - 1);
6872 } else {
6873 Register LaneValueLoReg =
6874 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6875 Register LaneValueHiReg =
6876 MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
6877 const TargetRegisterClass *SrcRC = MRI.getRegClass(SrcReg);
6878 MachineOperand FinalDPPResultOperand =
6879 MachineOperand::CreateReg(FinalDPPResult, /*isDef=*/false);
6880 auto [Op1L, Op1H] =
6881 ExtractSubRegs(MI, FinalDPPResultOperand, SrcRC, ST, MRI);
6882 // lane value input should be in an sgpr
6883 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_READLANE_B32),
6884 LaneValueLoReg)
6885 .addReg(Op1L)
6886 .addImm(ST.getWavefrontSize() - 1);
6887 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::V_READLANE_B32),
6888 LaneValueHiReg)
6889 .addReg(Op1H)
6890 .addImm(ST.getWavefrontSize() - 1);
6891 BuildRegSequence(*CurrBB, MI, ReducedValSGPR, LaneValueLoReg,
6892 LaneValueHiReg);
6893 }
6894 if (Opc == AMDGPU::S_SUB_I32) {
6895 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::S_SUB_I32), NegatedReducedVal)
6896 .addImm(0)
6897 .addReg(ReducedValSGPR)
6898 .setOperandDead(3); // Dead scc
6899 } else if (Opc == AMDGPU::S_SUB_U64_PSEUDO) {
6900 auto NegatedValInstr =
6901 BuildMI(*CurrBB, MI, DL, TII->get(Opc), NegatedReducedVal)
6902 .addImm(0)
6903 .addReg(ReducedValSGPR)
6904 .setOperandDead(3); // Dead scc
6905 CurrBB = expand64BitScalarArithmetic(*NegatedValInstr, CurrBB);
6906 }
6907 // Mark the final result as a whole-wave-mode calculation.
6908 BuildMI(*CurrBB, MI, DL, TII->get(AMDGPU::STRICT_WWM), DstReg)
6909 .addReg(Opc == AMDGPU::S_SUB_I32 || Opc == AMDGPU::S_SUB_U64_PSEUDO
6910 ? NegatedReducedVal
6911 : ReducedValSGPR);
6912 RetBB = CurrBB;
6913 }
6914 }
6915 MI.eraseFromParent();
6916 return RetBB;
6917}
6918
6921 MachineBasicBlock *BB) const {
6922 MachineFunction *MF = BB->getParent();
6924 const GCNSubtarget &ST = MF->getSubtarget<GCNSubtarget>();
6926 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
6927 MachineRegisterInfo &MRI = MF->getRegInfo();
6928 const DebugLoc &DL = MI.getDebugLoc();
6929
6930 switch (MI.getOpcode()) {
6931 case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U32:
6932 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MIN_U32);
6933 case AMDGPU::WAVE_REDUCE_UMIN_PSEUDO_U64:
6934 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_CMP_LT_U64_e64);
6935 case AMDGPU::WAVE_REDUCE_MIN_PSEUDO_I32:
6936 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MIN_I32);
6937 case AMDGPU::WAVE_REDUCE_MIN_PSEUDO_I64:
6938 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_CMP_LT_I64_e64);
6939 case AMDGPU::WAVE_REDUCE_FMIN_PSEUDO_F32:
6940 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_MIN_F32_e64);
6941 case AMDGPU::WAVE_REDUCE_FMIN_PSEUDO_F64:
6942 return lowerWaveReduce(MI, *BB, *getSubtarget(),
6943 ST.getGeneration() >= AMDGPUSubtarget::GFX12
6944 ? AMDGPU::V_MIN_NUM_F64_e64
6945 : AMDGPU::V_MIN_F64_e64);
6946 case AMDGPU::WAVE_REDUCE_UMAX_PSEUDO_U32:
6947 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MAX_U32);
6948 case AMDGPU::WAVE_REDUCE_UMAX_PSEUDO_U64:
6949 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_CMP_GT_U64_e64);
6950 case AMDGPU::WAVE_REDUCE_MAX_PSEUDO_I32:
6951 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_MAX_I32);
6952 case AMDGPU::WAVE_REDUCE_MAX_PSEUDO_I64:
6953 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_CMP_GT_I64_e64);
6954 case AMDGPU::WAVE_REDUCE_FMAX_PSEUDO_F32:
6955 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_MAX_F32_e64);
6956 case AMDGPU::WAVE_REDUCE_FMAX_PSEUDO_F64:
6957 return lowerWaveReduce(MI, *BB, *getSubtarget(),
6958 ST.getGeneration() >= AMDGPUSubtarget::GFX12
6959 ? AMDGPU::V_MAX_NUM_F64_e64
6960 : AMDGPU::V_MAX_F64_e64);
6961 case AMDGPU::WAVE_REDUCE_ADD_PSEUDO_I32:
6962 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_ADD_I32);
6963 case AMDGPU::WAVE_REDUCE_ADD_PSEUDO_U64:
6964 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_ADD_U64_PSEUDO);
6965 case AMDGPU::WAVE_REDUCE_FADD_PSEUDO_F32:
6966 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_ADD_F32_e64);
6967 case AMDGPU::WAVE_REDUCE_FADD_PSEUDO_F64:
6968 return lowerWaveReduce(MI, *BB, *getSubtarget(),
6969 ST.getGeneration() >= AMDGPUSubtarget::GFX12
6970 ? AMDGPU::V_ADD_F64_pseudo_e64
6971 : AMDGPU::V_ADD_F64_e64);
6972 case AMDGPU::WAVE_REDUCE_SUB_PSEUDO_I32:
6973 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_SUB_I32);
6974 case AMDGPU::WAVE_REDUCE_SUB_PSEUDO_U64:
6975 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_SUB_U64_PSEUDO);
6976 case AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F32:
6977 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::V_SUB_F32_e64);
6978 case AMDGPU::WAVE_REDUCE_FSUB_PSEUDO_F64:
6979 // There is no S/V_SUB_F64 opcode. Double type subtraction is expanded as
6980 // fadd + neg, by setting the NEG bit in the instruction.
6981 return lowerWaveReduce(MI, *BB, *getSubtarget(),
6982 ST.getGeneration() >= AMDGPUSubtarget::GFX12
6983 ? AMDGPU::V_ADD_F64_pseudo_e64
6984 : AMDGPU::V_ADD_F64_e64);
6985 case AMDGPU::WAVE_REDUCE_AND_PSEUDO_B32:
6986 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_AND_B32);
6987 case AMDGPU::WAVE_REDUCE_AND_PSEUDO_B64:
6988 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_AND_B64);
6989 case AMDGPU::WAVE_REDUCE_OR_PSEUDO_B32:
6990 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_OR_B32);
6991 case AMDGPU::WAVE_REDUCE_OR_PSEUDO_B64:
6992 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_OR_B64);
6993 case AMDGPU::WAVE_REDUCE_XOR_PSEUDO_B32:
6994 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_XOR_B32);
6995 case AMDGPU::WAVE_REDUCE_XOR_PSEUDO_B64:
6996 return lowerWaveReduce(MI, *BB, *getSubtarget(), AMDGPU::S_XOR_B64);
6997 case AMDGPU::S_UADDO_PSEUDO:
6998 case AMDGPU::S_USUBO_PSEUDO: {
6999 MachineOperand &Dest0 = MI.getOperand(0);
7000 MachineOperand &Dest1 = MI.getOperand(1);
7001 MachineOperand &Src0 = MI.getOperand(2);
7002 MachineOperand &Src1 = MI.getOperand(3);
7003
7004 unsigned Opc = (MI.getOpcode() == AMDGPU::S_UADDO_PSEUDO)
7005 ? AMDGPU::S_ADD_U32
7006 : AMDGPU::S_SUB_U32;
7007 // clang-format off
7008 BuildMI(*BB, MI, DL, TII->get(Opc), Dest0.getReg())
7009 .add(Src0)
7010 .add(Src1);
7011 // clang-format on
7012
7013 unsigned SelOpc =
7014 Subtarget->isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
7015 BuildMI(*BB, MI, DL, TII->get(SelOpc), Dest1.getReg()).addImm(-1).addImm(0);
7016
7017 MI.eraseFromParent();
7018 return BB;
7019 }
7020 case AMDGPU::S_ADD_U64_PSEUDO:
7021 case AMDGPU::S_SUB_U64_PSEUDO: {
7022 return expand64BitScalarArithmetic(MI, BB);
7023 }
7024 case AMDGPU::V_ADD_U64_PSEUDO:
7025 case AMDGPU::V_SUB_U64_PSEUDO: {
7026 bool IsAdd = (MI.getOpcode() == AMDGPU::V_ADD_U64_PSEUDO);
7027
7028 MachineOperand &Dest = MI.getOperand(0);
7029 MachineOperand &Src0 = MI.getOperand(1);
7030 MachineOperand &Src1 = MI.getOperand(2);
7031
7032 if (ST.hasAddSubU64Insts()) {
7033 auto I = BuildMI(*BB, MI, DL,
7034 TII->get(IsAdd ? AMDGPU::V_ADD_U64_e64
7035 : AMDGPU::V_SUB_U64_e64),
7036 Dest.getReg())
7037 .add(Src0)
7038 .add(Src1)
7039 .addImm(0); // clamp
7040 TII->legalizeOperands(*I);
7041 MI.eraseFromParent();
7042 return BB;
7043 }
7044
7045 if (IsAdd && ST.hasLshlAddU64Inst()) {
7046 auto Add = BuildMI(*BB, MI, DL, TII->get(AMDGPU::V_LSHL_ADD_U64_e64),
7047 Dest.getReg())
7048 .add(Src0)
7049 .addImm(0)
7050 .add(Src1);
7051 TII->legalizeOperands(*Add);
7052 MI.eraseFromParent();
7053 return BB;
7054 }
7055
7056 const auto *CarryRC = TRI->getWaveMaskRegClass();
7057
7058 Register DestSub0 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
7059 Register DestSub1 = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
7060
7061 Register CarryReg = MRI.createVirtualRegister(CarryRC);
7062 Register DeadCarryReg = MRI.createVirtualRegister(CarryRC);
7063
7064 const TargetRegisterClass *Src0RC = Src0.isReg()
7065 ? MRI.getRegClass(Src0.getReg())
7066 : &AMDGPU::VReg_64RegClass;
7067 const TargetRegisterClass *Src1RC = Src1.isReg()
7068 ? MRI.getRegClass(Src1.getReg())
7069 : &AMDGPU::VReg_64RegClass;
7070
7071 const TargetRegisterClass *Src0SubRC =
7072 TRI->getSubRegisterClass(Src0RC, AMDGPU::sub0);
7073 const TargetRegisterClass *Src1SubRC =
7074 TRI->getSubRegisterClass(Src1RC, AMDGPU::sub1);
7075
7076 MachineOperand SrcReg0Sub0 = TII->buildExtractSubRegOrImm(
7077 MI, MRI, Src0, Src0RC, AMDGPU::sub0, Src0SubRC);
7078 MachineOperand SrcReg1Sub0 = TII->buildExtractSubRegOrImm(
7079 MI, MRI, Src1, Src1RC, AMDGPU::sub0, Src1SubRC);
7080
7081 MachineOperand SrcReg0Sub1 = TII->buildExtractSubRegOrImm(
7082 MI, MRI, Src0, Src0RC, AMDGPU::sub1, Src0SubRC);
7083 MachineOperand SrcReg1Sub1 = TII->buildExtractSubRegOrImm(
7084 MI, MRI, Src1, Src1RC, AMDGPU::sub1, Src1SubRC);
7085
7086 unsigned LoOpc =
7087 IsAdd ? AMDGPU::V_ADD_CO_U32_e64 : AMDGPU::V_SUB_CO_U32_e64;
7088 MachineInstr *LoHalf = BuildMI(*BB, MI, DL, TII->get(LoOpc), DestSub0)
7089 .addReg(CarryReg, RegState::Define)
7090 .add(SrcReg0Sub0)
7091 .add(SrcReg1Sub0)
7092 .addImm(0); // clamp bit
7093
7094 unsigned HiOpc = IsAdd ? AMDGPU::V_ADDC_U32_e64 : AMDGPU::V_SUBB_U32_e64;
7095 MachineInstr *HiHalf =
7096 BuildMI(*BB, MI, DL, TII->get(HiOpc), DestSub1)
7097 .addReg(DeadCarryReg, RegState::Define | RegState::Dead)
7098 .add(SrcReg0Sub1)
7099 .add(SrcReg1Sub1)
7100 .addReg(CarryReg, RegState::Kill)
7101 .addImm(0); // clamp bit
7102
7103 BuildMI(*BB, MI, DL, TII->get(TargetOpcode::REG_SEQUENCE), Dest.getReg())
7104 .addReg(DestSub0)
7105 .addImm(AMDGPU::sub0)
7106 .addReg(DestSub1)
7107 .addImm(AMDGPU::sub1);
7108 TII->legalizeOperands(*LoHalf);
7109 TII->legalizeOperands(*HiHalf);
7110 MI.eraseFromParent();
7111 return BB;
7112 }
7113 case AMDGPU::S_ADD_CO_PSEUDO:
7114 case AMDGPU::S_SUB_CO_PSEUDO: {
7115 // This pseudo has a chance to be selected
7116 // only from uniform add/subcarry node. All the VGPR operands
7117 // therefore assumed to be splat vectors.
7119 MachineOperand &Dest = MI.getOperand(0);
7120 MachineOperand &CarryDest = MI.getOperand(1);
7121 MachineOperand &Src0 = MI.getOperand(2);
7122 MachineOperand &Src1 = MI.getOperand(3);
7123 MachineOperand &Src2 = MI.getOperand(4);
7124 if (Src0.isReg() && TRI->isVectorRegister(MRI, Src0.getReg())) {
7125 Register RegOp0 = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7126 BuildMI(*BB, MII, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp0)
7127 .addReg(Src0.getReg());
7128 Src0.setReg(RegOp0);
7129 }
7130 if (Src1.isReg() && TRI->isVectorRegister(MRI, Src1.getReg())) {
7131 Register RegOp1 = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7132 BuildMI(*BB, MII, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp1)
7133 .addReg(Src1.getReg());
7134 Src1.setReg(RegOp1);
7135 }
7136 Register RegOp2 = MRI.createVirtualRegister(&AMDGPU::SReg_32_XM0RegClass);
7137 if (TRI->isVectorRegister(MRI, Src2.getReg())) {
7138 BuildMI(*BB, MII, DL, TII->get(AMDGPU::V_READFIRSTLANE_B32), RegOp2)
7139 .addReg(Src2.getReg());
7140 Src2.setReg(RegOp2);
7141 }
7142
7143 if (ST.isWave64()) {
7144 if (ST.hasScalarCompareEq64()) {
7145 BuildMI(*BB, MII, DL, TII->get(AMDGPU::S_CMP_LG_U64))
7146 .addReg(Src2.getReg())
7147 .addImm(0);
7148 } else {
7149 const TargetRegisterClass *Src2RC = MRI.getRegClass(Src2.getReg());
7150 const TargetRegisterClass *SubRC =
7151 TRI->getSubRegisterClass(Src2RC, AMDGPU::sub0);
7152 MachineOperand Src2Sub0 = TII->buildExtractSubRegOrImm(
7153 MII, MRI, Src2, Src2RC, AMDGPU::sub0, SubRC);
7154 MachineOperand Src2Sub1 = TII->buildExtractSubRegOrImm(
7155 MII, MRI, Src2, Src2RC, AMDGPU::sub1, SubRC);
7156 Register Src2_32 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7157
7158 BuildMI(*BB, MII, DL, TII->get(AMDGPU::S_OR_B32), Src2_32)
7159 .add(Src2Sub0)
7160 .add(Src2Sub1);
7161
7162 BuildMI(*BB, MII, DL, TII->get(AMDGPU::S_CMP_LG_U32))
7163 .addReg(Src2_32, RegState::Kill)
7164 .addImm(0);
7165 }
7166 } else {
7167 BuildMI(*BB, MII, DL, TII->get(AMDGPU::S_CMP_LG_U32))
7168 .addReg(Src2.getReg())
7169 .addImm(0);
7170 }
7171
7172 unsigned Opc = MI.getOpcode() == AMDGPU::S_ADD_CO_PSEUDO
7173 ? AMDGPU::S_ADDC_U32
7174 : AMDGPU::S_SUBB_U32;
7175
7176 BuildMI(*BB, MII, DL, TII->get(Opc), Dest.getReg()).add(Src0).add(Src1);
7177
7178 unsigned SelOpc =
7179 ST.isWave64() ? AMDGPU::S_CSELECT_B64 : AMDGPU::S_CSELECT_B32;
7180
7181 BuildMI(*BB, MII, DL, TII->get(SelOpc), CarryDest.getReg())
7182 .addImm(-1)
7183 .addImm(0);
7184
7185 MI.eraseFromParent();
7186 return BB;
7187 }
7188 case AMDGPU::SI_INIT_M0: {
7189 MachineOperand &M0Init = MI.getOperand(0);
7190 BuildMI(*BB, MI.getIterator(), MI.getDebugLoc(),
7191 TII->get(M0Init.isReg() ? AMDGPU::COPY : AMDGPU::S_MOV_B32),
7192 AMDGPU::M0)
7193 .add(M0Init);
7194 MI.eraseFromParent();
7195 return BB;
7196 }
7197 case AMDGPU::S_BARRIER_SIGNAL_ISFIRST_IMM: {
7198 // Set SCC to true, in case the barrier instruction gets converted to a NOP.
7199 BuildMI(*BB, MI.getIterator(), MI.getDebugLoc(),
7200 TII->get(AMDGPU::S_CMP_EQ_U32))
7201 .addImm(0)
7202 .addImm(0);
7203 return BB;
7204 }
7205 case AMDGPU::GET_GROUPSTATICSIZE: {
7206 assert(getTargetMachine().getTargetTriple().getOS() == Triple::AMDHSA ||
7207 getTargetMachine().getTargetTriple().getOS() == Triple::AMDPAL);
7208 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_MOV_B32))
7209 .add(MI.getOperand(0))
7210 .addImm(MFI->getLDSSize());
7211 MI.eraseFromParent();
7212 return BB;
7213 }
7214 case AMDGPU::GET_SHADERCYCLESHILO: {
7215 assert(MF->getSubtarget<GCNSubtarget>().hasShaderCyclesHiLoRegisters());
7216 // The algorithm is:
7217 //
7218 // hi1 = getreg(SHADER_CYCLES_HI)
7219 // lo1 = getreg(SHADER_CYCLES_LO)
7220 // hi2 = getreg(SHADER_CYCLES_HI)
7221 //
7222 // If hi1 == hi2 then there was no overflow and the result is hi2:lo1.
7223 // Otherwise there was overflow and the result is hi2:0. In both cases the
7224 // result should represent the actual time at some point during the sequence
7225 // of three getregs.
7226 using namespace AMDGPU::Hwreg;
7227 Register RegHi1 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7228 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_GETREG_B32), RegHi1)
7229 .addImm(HwregEncoding::encode(ID_SHADER_CYCLES_HI, 0, 32));
7230 Register RegLo1 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7231 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_GETREG_B32), RegLo1)
7232 .addImm(HwregEncoding::encode(ID_SHADER_CYCLES, 0, 32));
7233 Register RegHi2 = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7234 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_GETREG_B32), RegHi2)
7235 .addImm(HwregEncoding::encode(ID_SHADER_CYCLES_HI, 0, 32));
7236 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_CMP_EQ_U32))
7237 .addReg(RegHi1)
7238 .addReg(RegHi2);
7239 Register RegLo = MRI.createVirtualRegister(&AMDGPU::SReg_32RegClass);
7240 BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_CSELECT_B32), RegLo)
7241 .addReg(RegLo1)
7242 .addImm(0);
7243 BuildMI(*BB, MI, DL, TII->get(AMDGPU::REG_SEQUENCE))
7244 .add(MI.getOperand(0))
7245 .addReg(RegLo)
7246 .addImm(AMDGPU::sub0)
7247 .addReg(RegHi2)
7248 .addImm(AMDGPU::sub1);
7249 MI.eraseFromParent();
7250 return BB;
7251 }
7252 case AMDGPU::SI_INDIRECT_SRC_V1:
7253 case AMDGPU::SI_INDIRECT_SRC_V2:
7254 case AMDGPU::SI_INDIRECT_SRC_V3:
7255 case AMDGPU::SI_INDIRECT_SRC_V4:
7256 case AMDGPU::SI_INDIRECT_SRC_V5:
7257 case AMDGPU::SI_INDIRECT_SRC_V6:
7258 case AMDGPU::SI_INDIRECT_SRC_V7:
7259 case AMDGPU::SI_INDIRECT_SRC_V8:
7260 case AMDGPU::SI_INDIRECT_SRC_V9:
7261 case AMDGPU::SI_INDIRECT_SRC_V10:
7262 case AMDGPU::SI_INDIRECT_SRC_V11:
7263 case AMDGPU::SI_INDIRECT_SRC_V12:
7264 case AMDGPU::SI_INDIRECT_SRC_V16:
7265 case AMDGPU::SI_INDIRECT_SRC_V32:
7266 return emitIndirectSrc(MI, *BB, *getSubtarget());
7267 case AMDGPU::SI_INDIRECT_DST_V1:
7268 case AMDGPU::SI_INDIRECT_DST_V2:
7269 case AMDGPU::SI_INDIRECT_DST_V3:
7270 case AMDGPU::SI_INDIRECT_DST_V4:
7271 case AMDGPU::SI_INDIRECT_DST_V5:
7272 case AMDGPU::SI_INDIRECT_DST_V6:
7273 case AMDGPU::SI_INDIRECT_DST_V7:
7274 case AMDGPU::SI_INDIRECT_DST_V8:
7275 case AMDGPU::SI_INDIRECT_DST_V9:
7276 case AMDGPU::SI_INDIRECT_DST_V10:
7277 case AMDGPU::SI_INDIRECT_DST_V11:
7278 case AMDGPU::SI_INDIRECT_DST_V12:
7279 case AMDGPU::SI_INDIRECT_DST_V16:
7280 case AMDGPU::SI_INDIRECT_DST_V32:
7281 return emitIndirectDst(MI, *BB, *getSubtarget());
7282 case AMDGPU::SI_KILL_F32_COND_IMM_PSEUDO:
7283 case AMDGPU::SI_KILL_I1_PSEUDO:
7284 return splitKillBlock(MI, BB);
7285 case AMDGPU::V_CNDMASK_B64_PSEUDO: {
7287 return BB;
7288 }
7289 case AMDGPU::SI_BR_UNDEF: {
7290 MachineInstr *Br = BuildMI(*BB, MI, DL, TII->get(AMDGPU::S_CBRANCH_SCC1))
7291 .add(MI.getOperand(0));
7292 Br->getOperand(1).setIsUndef(); // read undef SCC
7293 MI.eraseFromParent();
7294 return BB;
7295 }
7296 case AMDGPU::ADJCALLSTACKUP:
7297 case AMDGPU::ADJCALLSTACKDOWN: {
7299 MachineInstrBuilder MIB(*MF, &MI);
7300 MIB.addReg(Info->getStackPtrOffsetReg(), RegState::ImplicitDefine)
7301 .addReg(Info->getStackPtrOffsetReg(), RegState::Implicit);
7302 return BB;
7303 }
7304 case AMDGPU::SI_CALL_ISEL: {
7305 unsigned ReturnAddrReg = TII->getRegisterInfo().getReturnAddressReg(*MF);
7306
7308 MIB = BuildMI(*BB, MI, DL, TII->get(AMDGPU::SI_CALL))
7309 .addDef(ReturnAddrReg, RegState::Dead);
7310
7311 for (const MachineOperand &MO : MI.operands())
7312 MIB.add(MO);
7313
7314 MIB.cloneMemRefs(MI);
7315 MIB.setMIFlags(MI.getFlags());
7316 MI.eraseFromParent();
7317 return BB;
7318 }
7319 case AMDGPU::V_ADDC_U32_e32:
7320 case AMDGPU::V_SUBB_U32_e32:
7321 case AMDGPU::V_SUBBREV_U32_e32:
7322 // These instructions have an implicit use of vcc which counts towards the
7323 // constant bus limit.
7324 TII->legalizeOperands(MI);
7325 return BB;
7326 case AMDGPU::DS_GWS_INIT:
7327 case AMDGPU::DS_GWS_SEMA_BR:
7328 case AMDGPU::DS_GWS_BARRIER:
7329 case AMDGPU::DS_GWS_SEMA_V:
7330 case AMDGPU::DS_GWS_SEMA_P:
7331 case AMDGPU::DS_GWS_SEMA_RELEASE_ALL:
7332 // A s_waitcnt 0 is required to be the instruction immediately following.
7333 if (getSubtarget()->hasGWSAutoReplay()) {
7335 return BB;
7336 }
7337
7338 return emitGWSMemViolTestLoop(MI, BB);
7339 case AMDGPU::S_SETREG_B32: {
7340 // Try to optimize cases that only set the denormal mode or rounding mode.
7341 //
7342 // If the s_setreg_b32 fully sets all of the bits in the rounding mode or
7343 // denormal mode to a constant, we can use s_round_mode or s_denorm_mode
7344 // instead.
7345 //
7346 // FIXME: This could be predicates on the immediate, but tablegen doesn't
7347 // allow you to have a no side effect instruction in the output of a
7348 // sideeffecting pattern.
7349 auto [ID, Offset, Width] =
7350 AMDGPU::Hwreg::HwregEncoding::decode(MI.getOperand(1).getImm());
7351 if (ID != AMDGPU::Hwreg::ID_MODE)
7352 return BB;
7353
7354 const unsigned WidthMask = maskTrailingOnes<unsigned>(Width);
7355 const unsigned SetMask = WidthMask << Offset;
7356
7357 if (getSubtarget()->hasDenormModeInst()) {
7358 unsigned SetDenormOp = 0;
7359 unsigned SetRoundOp = 0;
7360
7361 // The dedicated instructions can only set the whole denorm or round mode
7362 // at once, not a subset of bits in either.
7363 if (SetMask ==
7365 // If this fully sets both the round and denorm mode, emit the two
7366 // dedicated instructions for these.
7367 SetRoundOp = AMDGPU::S_ROUND_MODE;
7368 SetDenormOp = AMDGPU::S_DENORM_MODE;
7369 } else if (SetMask == AMDGPU::Hwreg::FP_ROUND_MASK) {
7370 SetRoundOp = AMDGPU::S_ROUND_MODE;
7371 } else if (SetMask == AMDGPU::Hwreg::FP_DENORM_MASK) {
7372 SetDenormOp = AMDGPU::S_DENORM_MODE;
7373 }
7374
7375 if (SetRoundOp || SetDenormOp) {
7376 MachineInstr *Def = MRI.getVRegDef(MI.getOperand(0).getReg());
7377 if (Def && Def->isMoveImmediate() && Def->getOperand(1).isImm()) {
7378 unsigned ImmVal = Def->getOperand(1).getImm();
7379 if (SetRoundOp) {
7380 BuildMI(*BB, MI, MI.getDebugLoc(), TII->get(SetRoundOp))
7381 .addImm(ImmVal & 0xf);
7382
7383 // If we also have the denorm mode, get just the denorm mode bits.
7384 ImmVal >>= 4;
7385 }
7386
7387 if (SetDenormOp) {
7388 BuildMI(*BB, MI, MI.getDebugLoc(), TII->get(SetDenormOp))
7389 .addImm(ImmVal & 0xf);
7390 }
7391
7392 MI.eraseFromParent();
7393 return BB;
7394 }
7395 }
7396 }
7397
7398 // If only FP bits are touched, used the no side effects pseudo.
7399 if ((SetMask & (AMDGPU::Hwreg::FP_ROUND_MASK |
7400 AMDGPU::Hwreg::FP_DENORM_MASK)) == SetMask)
7401 MI.setDesc(TII->get(AMDGPU::S_SETREG_B32_mode));
7402
7403 return BB;
7404 }
7405 case AMDGPU::S_INVERSE_BALLOT_U32:
7406 case AMDGPU::S_INVERSE_BALLOT_U64:
7407 // These opcodes only exist to let SIFixSGPRCopies insert a readfirstlane if
7408 // necessary. After that they are equivalent to a COPY.
7409 MI.setDesc(TII->get(AMDGPU::COPY));
7410 return BB;
7411 case AMDGPU::ENDPGM_TRAP: {
7412 if (BB->succ_empty() && std::next(MI.getIterator()) == BB->end()) {
7413 MI.setDesc(TII->get(AMDGPU::S_ENDPGM));
7414 MI.addOperand(MachineOperand::CreateImm(0));
7415 return BB;
7416 }
7417
7418 // We need a block split to make the real endpgm a terminator. We also don't
7419 // want to break phis in successor blocks, so we can't just delete to the
7420 // end of the block.
7421
7422 MachineBasicBlock *SplitBB = BB->splitAt(MI, false /*UpdateLiveIns*/);
7424 MF->push_back(TrapBB);
7425 // clang-format off
7426 BuildMI(*TrapBB, TrapBB->end(), DL, TII->get(AMDGPU::S_ENDPGM))
7427 .addImm(0);
7428 BuildMI(*BB, &MI, DL, TII->get(AMDGPU::S_CBRANCH_EXECNZ))
7429 .addMBB(TrapBB);
7430 // clang-format on
7431
7432 BB->addSuccessor(TrapBB);
7433 MI.eraseFromParent();
7434 return SplitBB;
7435 }
7436 case AMDGPU::SIMULATED_TRAP: {
7437 assert(Subtarget->hasPrivEnabledTrap2NopBug());
7438 MachineBasicBlock *SplitBB =
7439 TII->insertSimulatedTrap(MRI, *BB, MI, MI.getDebugLoc());
7440 MI.eraseFromParent();
7441 return SplitBB;
7442 }
7443 case AMDGPU::SI_TCRETURN_GFX_WholeWave:
7444 case AMDGPU::SI_WHOLE_WAVE_FUNC_RETURN: {
7446
7447 // During ISel, it's difficult to propagate the original EXEC mask to use as
7448 // an input to SI_WHOLE_WAVE_FUNC_RETURN. Set it up here instead.
7449 MachineInstr *Setup = TII->getWholeWaveFunctionSetup(*BB->getParent());
7450 assert(Setup && "Couldn't find SI_SETUP_WHOLE_WAVE_FUNC");
7451 Register OriginalExec = Setup->getOperand(0).getReg();
7452 MF->getRegInfo().clearKillFlags(OriginalExec);
7453 MI.getOperand(0).setReg(OriginalExec);
7454 return BB;
7455 }
7456 case AMDGPU::V_DOT2_F32_F16:
7457 case AMDGPU::V_DOT2_F32_BF16: {
7458 // Hint RA to assign dst and src2 the same physical register.
7459 // For targets without VOP2, but with VOPD, variant of the instruction this
7460 // is one of the conditions to attempt converting VOP3P to VOPD.
7461 MRI.setSimpleHint(MI.getOperand(0).getReg(), MI.getOperand(6).getReg());
7462 return BB;
7463 }
7464 case AMDGPU::SCHED_BARRIER:
7465 case AMDGPU::SCHED_GROUP_BARRIER:
7466 MI.getOperand(0).setImm(MI.getOperand(0).getImm() &
7467 static_cast<unsigned>(AMDGPU::SchedGroupMask::ALL));
7468 return BB;
7469 default:
7470 if (TII->isImage(MI) || TII->isMUBUF(MI)) {
7471 if (!MI.mayStore())
7473 return BB;
7474 }
7476 }
7477}
7478
7480 // This currently forces unfolding various combinations of fsub into fma with
7481 // free fneg'd operands. As long as we have fast FMA (controlled by
7482 // isFMAFasterThanFMulAndFAdd), we should perform these.
7483
7484 // When fma is quarter rate, for f64 where add / sub are at best half rate,
7485 // most of these combines appear to be cycle neutral but save on instruction
7486 // count / code size.
7487 return true;
7488}
7489
7491
7493 EVT VT) const {
7494 if (!VT.isVector()) {
7495 return MVT::i1;
7496 }
7497 return EVT::getVectorVT(Ctx, MVT::i1, VT.getVectorNumElements());
7498}
7499
7501 // TODO: Should i16 be used always if legal? For now it would force VALU
7502 // shifts.
7503 return (VT == MVT::i16) ? MVT::i16 : MVT::i32;
7504}
7505
7507 return (Ty.getScalarSizeInBits() <= 16 && Subtarget->has16BitInsts())
7508 ? Ty.changeElementSize(16)
7509 : Ty.changeElementSize(32);
7510}
7511
7512// Answering this is somewhat tricky and depends on the specific device which
7513// have different rates for fma or all f64 operations.
7514//
7515// v_fma_f64 and v_mul_f64 always take the same number of cycles as each other
7516// regardless of which device (although the number of cycles differs between
7517// devices), so it is always profitable for f64.
7518//
7519// v_fma_f32 takes 4 or 16 cycles depending on the device, so it is profitable
7520// only on full rate devices. Normally, we should prefer selecting v_mad_f32
7521// which we can always do even without fused FP ops since it returns the same
7522// result as the separate operations and since it is always full
7523// rate. Therefore, we lie and report that it is not faster for f32. v_mad_f32
7524// however does not support denormals, so we do report fma as faster if we have
7525// a fast fma device and require denormals.
7526//
7528 DenormalFPEnv FPEnv) const {
7529 VT = VT.getScalarType();
7530 if (!VT.isSimple())
7531 return false;
7532
7533 switch (VT.getSimpleVT().SimpleTy) {
7534 case MVT::f32: {
7535 // If mad is not available this depends only on if f32 fma is full rate.
7536 if (!Subtarget->hasMadMacF32Insts())
7537 return Subtarget->hasFastFMAF32();
7538
7539 // Otherwise f32 mad is always full rate and returns the same result as
7540 // the separate operations so should be preferred over fma.
7541 // However does not support denormals.
7543 return Subtarget->hasFastFMAF32() || Subtarget->hasDLInsts();
7544
7545 // If the subtarget has v_fmac_f32, that's just as good as v_mac_f32.
7546 return Subtarget->hasFastFMAF32() && Subtarget->hasDLInsts();
7547 }
7548 case MVT::f64:
7549 return true;
7550 case MVT::f16:
7551 case MVT::bf16:
7552 return Subtarget->has16BitInsts() &&
7554 default:
7555 break;
7556 }
7557
7558 return false;
7559}
7560
7565
7567 Type *Ty) const {
7569 getValueType(F.getDataLayout(), Ty, /*AllowUnknown=*/true),
7570 F.getDenormalFPEnv());
7571}
7572
7574 LLT Ty) const {
7575 switch (Ty.getScalarSizeInBits()) {
7576 case 16:
7577 return isFMAFasterThanFMulAndFAdd(MF, MVT::f16);
7578 case 32:
7579 return isFMAFasterThanFMulAndFAdd(MF, MVT::f32);
7580 case 64:
7581 return isFMAFasterThanFMulAndFAdd(MF, MVT::f64);
7582 default:
7583 break;
7584 }
7585
7586 return false;
7587}
7588
7590 // TODO: Check future ftz flag
7591 // v_mad_f32/v_mac_f32 do not support denormals.
7592 if (VT == MVT::f32)
7593 return Subtarget->hasMadMacF32Insts() &&
7595 if (VT == MVT::f16)
7596 return Subtarget->hasMadF16() &&
7598
7599 return false;
7600}
7601
7603 if (!Ty.isScalar())
7604 return false;
7605
7606 DenormalFPEnv FPEnv = getDenormalFPEnv(*MI.getMF());
7607 if (Ty.getScalarSizeInBits() == 16)
7608 return isFMADLegal(MVT::f16, FPEnv);
7609 if (Ty.getScalarSizeInBits() == 32)
7610 return isFMADLegal(MVT::f32, FPEnv);
7611
7612 return false;
7613}
7614
7616 const SDNode *N) const {
7617 return isFMADLegal(N->getValueType(0),
7619}
7620
7622 return isFMADLegal(getValueType(F.getDataLayout(), Ty->getScalarType(),
7623 /*AllowUnknown=*/true),
7624 F.getDenormalFPEnv());
7625}
7626
7627//===----------------------------------------------------------------------===//
7628// Custom DAG Lowering Operations
7629//===----------------------------------------------------------------------===//
7630
7631// Work around LegalizeDAG doing the wrong thing and fully scalarizing if the
7632// wider vector type is legal.
7634 SelectionDAG &DAG) const {
7635 unsigned Opc = Op.getOpcode();
7636 EVT VT = Op.getValueType();
7638
7639 auto [Lo, Hi] = DAG.SplitVectorOperand(Op.getNode(), 0);
7640 auto [LoVT, HiVT] = DAG.GetSplitDestVTs(VT);
7641
7642 SDLoc SL(Op);
7643
7644 // Forward any trailing scalar operands unchanged to both halves.
7645 SmallVector<SDValue, 2> LoOps = {Lo};
7646 SmallVector<SDValue, 2> HiOps = {Hi};
7647 auto TrailingOps = drop_begin(Op->ops());
7648 LoOps.append(TrailingOps.begin(), TrailingOps.end());
7649 HiOps.append(TrailingOps.begin(), TrailingOps.end());
7650
7651 SDValue OpLo = DAG.getNode(Opc, SL, LoVT, LoOps, Op->getFlags());
7652 SDValue OpHi = DAG.getNode(Opc, SL, HiVT, HiOps, Op->getFlags());
7653
7654 return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(Op), VT, OpLo, OpHi);
7655}
7656
7657// Enable lowering of ROTR for vxi32 types. This is a workaround for a
7658// regression whereby extra unnecessary instructions were added to codegen
7659// for rotr operations, casued by legalising v2i32 or. This resulted in extra
7660// instructions to extract the result from the vector.
7662 [[maybe_unused]] EVT VT = Op.getValueType();
7663
7664 assert((VT == MVT::v2i32 || VT == MVT::v4i32 || VT == MVT::v8i32 ||
7665 VT == MVT::v16i32) &&
7666 "Unexpected ValueType.");
7667
7668 return DAG.UnrollVectorOp(Op.getNode());
7669}
7670
7671// Work around LegalizeDAG doing the wrong thing and fully scalarizing if the
7672// wider vector type is legal.
7674 SelectionDAG &DAG) const {
7675 unsigned Opc = Op.getOpcode();
7676 EVT VT = Op.getValueType();
7678
7679 auto [Lo0, Hi0] = DAG.SplitVectorOperand(Op.getNode(), 0);
7680 auto [Lo1, Hi1] = DAG.SplitVectorOperand(Op.getNode(), 1);
7681
7682 SDLoc SL(Op);
7683
7684 SDValue OpLo =
7685 DAG.getNode(Opc, SL, Lo0.getValueType(), Lo0, Lo1, Op->getFlags());
7686 SDValue OpHi =
7687 DAG.getNode(Opc, SL, Hi0.getValueType(), Hi0, Hi1, Op->getFlags());
7688
7689 return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(Op), VT, OpLo, OpHi);
7690}
7691
7693 SelectionDAG &DAG) const {
7694 unsigned Opc = Op.getOpcode();
7695 EVT VT = Op.getValueType();
7697
7698 SDValue Op0 = Op.getOperand(0);
7699 SDValue Lo0, Hi0;
7700 if (Op0.getValueType().isVector())
7701 std::tie(Lo0, Hi0) = DAG.SplitVectorOperand(Op.getNode(), 0);
7702 else
7703 Lo0 = Hi0 = DAG.getFreeze(Op0);
7704
7705 auto [Lo1, Hi1] = DAG.SplitVectorOperand(Op.getNode(), 1);
7706 auto [Lo2, Hi2] = DAG.SplitVectorOperand(Op.getNode(), 2);
7707
7708 SDLoc SL(Op);
7709 auto ResVT = DAG.GetSplitDestVTs(VT);
7710
7711 SDValue OpLo =
7712 DAG.getNode(Opc, SL, ResVT.first, Lo0, Lo1, Lo2, Op->getFlags());
7713 SDValue OpHi =
7714 DAG.getNode(Opc, SL, ResVT.second, Hi0, Hi1, Hi2, Op->getFlags());
7715
7716 return DAG.getNode(ISD::CONCAT_VECTORS, SDLoc(Op), VT, OpLo, OpHi);
7717}
7718
7720 switch (Op.getOpcode()) {
7721 default:
7723 case ISD::BRCOND:
7724 return LowerBRCOND(Op, DAG);
7725 case ISD::RETURNADDR:
7726 return LowerRETURNADDR(Op, DAG);
7727 case ISD::SPONENTRY:
7728 return LowerSPONENTRY(Op, DAG);
7729 case ISD::LOAD: {
7730 SDValue Result = LowerLOAD(Op, DAG);
7731 assert((!Result.getNode() || Result.getNode()->getNumValues() == 2) &&
7732 "Load should return a value and a chain");
7733 return Result;
7734 }
7735 case ISD::FSQRT: {
7736 EVT VT = Op.getValueType();
7737 if (VT == MVT::f32)
7738 return lowerFSQRTF32(Op, DAG);
7739 if (VT == MVT::f64)
7740 return lowerFSQRTF64(Op, DAG);
7741 return SDValue();
7742 }
7743 case ISD::FSIN:
7744 case ISD::FCOS:
7745 return LowerTrig(Op, DAG);
7746 case ISD::SELECT:
7747 return LowerSELECT(Op, DAG);
7748 case ISD::FDIV:
7749 return LowerFDIV(Op, DAG);
7750 case ISD::FFREXP:
7751 return LowerFFREXP(Op, DAG);
7753 return LowerATOMIC_CMP_SWAP(Op, DAG);
7754 case ISD::STORE:
7755 return LowerSTORE(Op, DAG);
7756 case ISD::GlobalAddress: {
7759 return LowerGlobalAddress(MFI, Op, DAG);
7760 }
7761 case ISD::BlockAddress:
7762 return LowerBlockAddress(Op, DAG);
7764 return LowerExternalSymbol(Op, DAG);
7766 return LowerINTRINSIC_WO_CHAIN(Op, DAG);
7768 return LowerCONVERT_FROM_ARBITRARY_FP(Op, DAG);
7770 return LowerCONVERT_TO_ARBITRARY_FP(Op, DAG);
7772 return LowerINTRINSIC_W_CHAIN(Op, DAG);
7774 return LowerINTRINSIC_VOID(Op, DAG);
7775 case ISD::ADDRSPACECAST:
7776 return lowerADDRSPACECAST(Op, DAG);
7778 return lowerINSERT_SUBVECTOR(Op, DAG);
7780 return lowerINSERT_VECTOR_ELT(Op, DAG);
7782 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
7784 return lowerVECTOR_SHUFFLE(Op, DAG);
7786 return lowerSCALAR_TO_VECTOR(Op, DAG);
7787 case ISD::BUILD_VECTOR:
7788 return lowerBUILD_VECTOR(Op, DAG);
7789 case ISD::FP_ROUND:
7791 return lowerFP_ROUND(Op, DAG);
7792 case ISD::TRAP:
7793 return lowerTRAP(Op, DAG);
7794 case ISD::DEBUGTRAP:
7795 return lowerDEBUGTRAP(Op, DAG);
7796 case ISD::ABS:
7797 case ISD::FABS:
7798 case ISD::FNEG:
7799 case ISD::FCANONICALIZE:
7800 case ISD::BSWAP:
7801 return splitUnaryVectorOp(Op, DAG);
7804 if (Op.getValueType().isVector() && Op.getValueType() != MVT::v2i16 &&
7805 Op.getOperand(0).getValueType().getScalarType() == MVT::f32)
7806 return splitUnaryVectorOp(Op, DAG);
7807 return LowerFP_TO_INT_SAT(Op, DAG);
7808 case ISD::FSUB:
7809 if (Op.getValueType() == MVT::bf16) {
7810 // Custom expansion:
7811 // fsub bf16 %a, %b -> fadd v2bf16(widen %a), fneg v2bf16(widen %b)
7812 // Then extract back to bf16.
7813 //
7814 // We create fneg on v2bf16 (not bf16) so the instruction selector can
7815 // fold the negation into the packed add's neg_lo/neg_hi modifiers,
7816 // generating a single v_pk_add_bf16 instruction. If we negate bf16 first,
7817 // it becomes a separate v_xor instruction before widening.
7818 SDLoc DL(Op);
7819 SDValue Op0 = Op.getOperand(0);
7820 SDValue Op1 = Op.getOperand(1);
7821
7822 // Widen both operands to v2bf16
7823 SDValue Vec0 = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2bf16, Op0);
7824 SDValue Vec1 = DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v2bf16, Op1);
7825
7826 // Create FNEG v2bf16 for the second operand
7827 SDValue NegVec1 = DAG.getNode(ISD::FNEG, DL, MVT::v2bf16, Vec1);
7828
7829 // Perform FADD v2bf16
7830 SDValue Result = DAG.getNode(ISD::FADD, DL, MVT::v2bf16, Vec0, NegVec1);
7831
7832 // Extract element 0 back to bf16
7833 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::bf16, Result,
7834 DAG.getConstant(0, DL, MVT::i32));
7835 }
7836 return SDValue();
7837 case ISD::FMINNUM:
7838 case ISD::FMAXNUM:
7839 return lowerFMINNUM_FMAXNUM(Op, DAG);
7840 case ISD::FMINIMUMNUM:
7841 case ISD::FMAXIMUMNUM:
7842 return lowerFMINIMUMNUM_FMAXIMUMNUM(Op, DAG);
7843 case ISD::FLDEXP:
7844 case ISD::STRICT_FLDEXP:
7845 return lowerFLDEXP(Op, DAG);
7846 case ISD::FMA:
7847 return splitTernaryVectorOp(Op, DAG);
7848 case ISD::FP_TO_SINT:
7849 case ISD::FP_TO_UINT:
7850 if (Subtarget->hasVCvtPkIU16F32() && Op.getValueType() == MVT::i16 &&
7851 Op.getOperand(0).getValueType() == MVT::f32) {
7852 // Make f32->i16 legal so we can select V_CVT_PK_[IU]16_F32.
7853 return Op;
7854 }
7855 return LowerFP_TO_INT(Op, DAG);
7856 case ISD::SHL:
7857 case ISD::SRA:
7858 case ISD::SRL:
7859 case ISD::ADD:
7860 case ISD::SUB:
7861 case ISD::SMIN:
7862 case ISD::SMAX:
7863 case ISD::UMIN:
7864 case ISD::UMAX:
7865 case ISD::FMINNUM_IEEE:
7866 case ISD::FMAXNUM_IEEE:
7867 case ISD::FMINIMUM:
7868 case ISD::FMAXIMUM:
7869 case ISD::UADDSAT:
7870 case ISD::USUBSAT:
7871 case ISD::SADDSAT:
7872 case ISD::SSUBSAT:
7873 case ISD::FADD:
7874 case ISD::FMUL:
7875 return splitBinaryVectorOp(Op, DAG);
7876 case ISD::FCOPYSIGN:
7877 return lowerFCOPYSIGN(Op, DAG);
7878 case ISD::MUL:
7879 return lowerMUL(Op, DAG);
7880 case ISD::SMULO:
7881 case ISD::UMULO:
7882 return lowerXMULO(Op, DAG);
7883 case ISD::SMUL_LOHI:
7884 case ISD::UMUL_LOHI:
7885 return lowerXMUL_LOHI(Op, DAG);
7887 return LowerDYNAMIC_STACKALLOC(Op, DAG);
7888 case ISD::STACKSAVE:
7889 return LowerSTACKSAVE(Op, DAG);
7890 case ISD::GET_ROUNDING:
7891 return lowerGET_ROUNDING(Op, DAG);
7892 case ISD::SET_ROUNDING:
7893 return lowerSET_ROUNDING(Op, DAG);
7894 case ISD::PREFETCH:
7895 return lowerPREFETCH(Op, DAG);
7896 case ISD::FP_EXTEND:
7898 return lowerFP_EXTEND(Op, DAG);
7899 case ISD::GET_FPENV:
7900 return lowerGET_FPENV(Op, DAG);
7901 case ISD::SET_FPENV:
7902 return lowerSET_FPENV(Op, DAG);
7903 case ISD::ROTR:
7904 return lowerROTR(Op, DAG);
7905 case ISD::INLINEASM:
7906 return LowerINLINEASM(Op, DAG);
7907 }
7908 return SDValue();
7909}
7910
7911// TFE results are dword granular: value dwords followed by one status dword.
7912static std::pair<SDValue, SDValue>
7914 LLVMContext &C = *DAG.getContext();
7915 unsigned NumValueDWords = divideCeil(VT.getSizeInBits(), 32);
7917 DAG.getVectorIdxConstant(NumValueDWords, DL));
7918 SDValue ZeroIdx = DAG.getVectorIdxConstant(0, DL);
7919 SDValue ValueDWords =
7920 NumValueDWords == 1
7921 ? DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, Op, ZeroIdx)
7923 EVT::getVectorVT(C, MVT::i32, NumValueDWords), Op,
7924 ZeroIdx);
7925 if (!VT.isVector() && VT.getSizeInBits() < 32)
7926 ValueDWords =
7927 DAG.getNode(ISD::TRUNCATE, DL, VT.changeTypeToInteger(), ValueDWords);
7928 return {DAG.getNode(ISD::BITCAST, DL, VT, ValueDWords), Status};
7929}
7930
7931// Used for D16: Casts the result of an instruction into the right vector,
7932// packs values if loads return unpacked values.
7934 const SDLoc &DL, SelectionDAG &DAG,
7935 bool Unpacked) {
7936 if (!LoadVT.isVector())
7937 return Result;
7938
7939 // Cast back to the original packed type or to a larger type that is a
7940 // multiple of 32 bit for D16. Widening the return type is a required for
7941 // legalization.
7942 EVT FittingLoadVT = LoadVT;
7943 if ((LoadVT.getVectorNumElements() % 2) == 1) {
7944 FittingLoadVT =
7946 LoadVT.getVectorNumElements() + 1);
7947 }
7948
7949 if (Unpacked) { // From v2i32/v4i32 back to v2f16/v4f16.
7950 // Truncate to v2i16/v4i16.
7951 EVT IntLoadVT = FittingLoadVT.changeTypeToInteger();
7952
7953 // Workaround legalizer not scalarizing truncate after vector op
7954 // legalization but not creating intermediate vector trunc.
7956 DAG.ExtractVectorElements(Result, Elts);
7957 for (SDValue &Elt : Elts)
7958 Elt = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, Elt);
7959
7960 // Pad illegal v1i16/v3fi6 to v4i16
7961 if ((LoadVT.getVectorNumElements() % 2) == 1)
7962 Elts.push_back(DAG.getPOISON(MVT::i16));
7963
7964 Result = DAG.getBuildVector(IntLoadVT, DL, Elts);
7965
7966 // Bitcast to original type (v2f16/v4f16).
7967 return DAG.getNode(ISD::BITCAST, DL, FittingLoadVT, Result);
7968 }
7969
7970 // Cast back to the original packed type.
7971 return DAG.getNode(ISD::BITCAST, DL, FittingLoadVT, Result);
7972}
7973
7974SDValue SITargetLowering::adjustLoadValueType(unsigned Opcode, MemSDNode *M,
7975 SelectionDAG &DAG,
7977 bool IsIntrinsic) const {
7978 SDLoc DL(M);
7979
7980 bool IsTFE = M->getNumValues() == 3;
7981 bool Unpacked = Subtarget->hasUnpackedD16VMem();
7982 EVT LoadVT = M->getValueType(0);
7983
7984 EVT EquivLoadVT = LoadVT;
7985 if (LoadVT.isVector()) {
7986 if (Unpacked) {
7987 EquivLoadVT = EVT::getVectorVT(*DAG.getContext(), MVT::i32,
7988 LoadVT.getVectorNumElements());
7989 } else if ((LoadVT.getVectorNumElements() % 2) == 1) {
7990 // Widen v3f16 to legal type
7991 EquivLoadVT =
7993 LoadVT.getVectorNumElements() + 1);
7994 }
7995 }
7996
7997 if (IsTFE) {
7998 unsigned NumValueDWords = divideCeil(EquivLoadVT.getSizeInBits(), 32);
7999 EVT LoadDWordsVT =
8000 EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumValueDWords + 1);
8001 SDVTList VTList = DAG.getVTList(LoadDWordsVT, MVT::Other);
8002 SDValue Load = DAG.getMemIntrinsicNode(
8003 Opcode, DL, VTList, Ops, M->getMemoryVT(), M->getMemOperand());
8004 auto [Value, Status] = splitTFEValueAndStatus(Load, EquivLoadVT, DL, DAG);
8005 SDValue Adjusted =
8006 adjustLoadValueTypeImpl(Value, LoadVT, DL, DAG, Unpacked);
8007 return DAG.getMergeValues({Adjusted, Status, Load.getValue(1)}, DL);
8008 }
8009
8010 // Change from v4f16/v2f16 to EquivLoadVT.
8011 SDVTList VTList = DAG.getVTList(EquivLoadVT, MVT::Other);
8012
8013 SDValue Load = DAG.getMemIntrinsicNode(
8014 IsIntrinsic ? (unsigned)ISD::INTRINSIC_W_CHAIN : Opcode, DL, VTList, Ops,
8015 M->getMemoryVT(), M->getMemOperand());
8016
8017 SDValue Adjusted = adjustLoadValueTypeImpl(Load, LoadVT, DL, DAG, Unpacked);
8018
8019 return DAG.getMergeValues({Adjusted, Load.getValue(1)}, DL);
8020}
8021
8022SDValue SITargetLowering::lowerIntrinsicLoad(MemSDNode *M, bool IsFormat,
8023 SelectionDAG &DAG,
8024 ArrayRef<SDValue> Ops) const {
8025 SDLoc DL(M);
8026 EVT LoadVT = M->getValueType(0);
8027 EVT EltType = LoadVT.getScalarType();
8028 EVT IntVT = LoadVT.changeTypeToInteger();
8029
8030 bool IsD16 = IsFormat && (EltType.getSizeInBits() == 16);
8031
8032 if (IsFormat && !IsD16 && EltType.getSizeInBits() < 32) {
8033 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
8035 "unsupported sub-dword format buffer load", DL.getDebugLoc()));
8036 return DAG.getMergeValues({DAG.getPOISON(LoadVT), M->getOperand(0)}, DL);
8037 }
8038
8039 assert(M->getNumValues() == 2 || M->getNumValues() == 3);
8040 bool IsTFE = M->getNumValues() == 3;
8041
8042 if (IsD16 && IsTFE && !Subtarget->hasBufferTFEFormatD16()) {
8043 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
8045 "TFE D16 format buffer load is not supported on this GPU",
8046 DL.getDebugLoc()));
8047 return DAG.getErrorMergeValues({M->value_begin(), M->value_end()},
8048 M->getOperand(0), DL);
8049 }
8050
8051 unsigned Opc = IsD16 ? (IsTFE ? AMDGPUISD::BUFFER_LOAD_FORMAT_D16_TFE
8052 : AMDGPUISD::BUFFER_LOAD_FORMAT_D16)
8053 : IsFormat ? (IsTFE ? AMDGPUISD::BUFFER_LOAD_FORMAT_TFE
8054 : AMDGPUISD::BUFFER_LOAD_FORMAT)
8055 : IsTFE ? AMDGPUISD::BUFFER_LOAD_TFE
8056 : AMDGPUISD::BUFFER_LOAD;
8057
8058 if (IsD16)
8059 return adjustLoadValueType(Opc, M, DAG, Ops);
8060
8061 // Handle BUFFER_LOAD_BYTE/UBYTE/SHORT/USHORT overloaded intrinsics
8062 if (!IsD16 && !LoadVT.isVector() && EltType.getSizeInBits() < 32)
8063 return handleByteShortBufferLoads(DAG, LoadVT, DL, Ops, M->getMemOperand(),
8064 IsTFE);
8065
8066 if (isTypeLegal(LoadVT)) {
8067 return getMemIntrinsicNode(Opc, DL, M->getVTList(), Ops, IntVT,
8068 M->getMemOperand(), DAG);
8069 }
8070
8071 EVT CastVT = getEquivalentMemType(*DAG.getContext(), LoadVT);
8072 SDVTList VTList = IsTFE ? DAG.getVTList(CastVT, MVT::i32, MVT::Other)
8073 : DAG.getVTList(CastVT, MVT::Other);
8074 SDValue MemNode = getMemIntrinsicNode(Opc, DL, VTList, Ops, CastVT,
8075 M->getMemOperand(), DAG);
8076 SDValue Data = DAG.getNode(ISD::BITCAST, DL, LoadVT, MemNode);
8077 if (IsTFE)
8078 return DAG.getMergeValues({Data, MemNode.getValue(1), MemNode.getValue(2)},
8079 DL);
8080 return DAG.getMergeValues({Data, MemNode.getValue(1)}, DL);
8081}
8082
8084 SelectionDAG &DAG) {
8085 EVT VT = N->getValueType(0);
8086 SDValue Src = N->getOperand(1);
8087 SDLoc SL(N);
8088
8089 if (Src.getOpcode() == ISD::SETCC) {
8090 SDValue Op0 = Src.getOperand(0);
8091 SDValue Op1 = Src.getOperand(1);
8092 // Need to expand bfloat to float for comparison (setcc).
8093 if (Op0.getValueType() == MVT::bf16) {
8094 Op0 = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Op0);
8095 Op1 = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Op1);
8096 }
8097 // (ballot (ISD::SETCC ...)) -> (AMDGPUISD::SETCC ...)
8098 return DAG.getNode(AMDGPUISD::SETCC, SL, VT, Op0, Op1, Src.getOperand(2));
8099 }
8100 if (const ConstantSDNode *Arg = dyn_cast<ConstantSDNode>(Src)) {
8101 // (ballot 0) -> 0
8102 if (Arg->isZero())
8103 return DAG.getConstant(0, SL, VT);
8104
8105 // (ballot 1) -> EXEC/EXEC_LO
8106 if (Arg->isOne()) {
8107 Register Exec;
8108 if (VT.getScalarSizeInBits() == 32)
8109 Exec = AMDGPU::EXEC_LO;
8110 else if (VT.getScalarSizeInBits() == 64)
8111 Exec = AMDGPU::EXEC;
8112 else
8113 return SDValue();
8114
8115 return DAG.getCopyFromReg(DAG.getEntryNode(), SL, Exec, VT);
8116 }
8117 }
8118
8119 // (ballot (i1 $src)) -> (AMDGPUISD::SETCC (i32 (zext $src)) (i32 0)
8120 // ISD::SETNE)
8121 return DAG.getNode(
8122 AMDGPUISD::SETCC, SL, VT, DAG.getZExtOrTrunc(Src, SL, MVT::i32),
8123 DAG.getConstant(0, SL, MVT::i32), DAG.getCondCode(ISD::SETNE));
8124}
8125
8127 Intrinsic::ID IntrinsicID) {
8128 bool Signed = IntrinsicID == Intrinsic::amdgcn_sbfe;
8129 SDLoc DL(Op);
8130 EVT VT = Op.getValueType();
8131 SDValue Src = Op.getOperand(1);
8132 SDValue Offset = Op.getOperand(2);
8133 SDValue Width = Op.getOperand(3);
8134
8135 if (VT != MVT::i32) {
8138 Twine(Intrinsic::getBaseName(IntrinsicID)) + " only supports i32",
8139 DL.getDebugLoc()));
8140 return DAG.getPOISON(VT);
8141 }
8142
8143 return DAG.getNode(Signed ? AMDGPUISD::BFE_I32 : AMDGPUISD::BFE_U32, DL, VT,
8144 Src, Offset, Width);
8145}
8146
8148 EVT VT);
8149
8151 SelectionDAG &DAG) {
8152 EVT VT = N->getValueType(0);
8153 unsigned ValSize = VT.getSizeInBits();
8154 unsigned IID = N->getConstantOperandVal(0);
8155 bool IsPermLane16 = IID == Intrinsic::amdgcn_permlane16 ||
8156 IID == Intrinsic::amdgcn_permlanex16;
8157 bool IsSetInactive = IID == Intrinsic::amdgcn_set_inactive ||
8158 IID == Intrinsic::amdgcn_set_inactive_chain_arg;
8159 bool IsPermlaneShuffle = IID == Intrinsic::amdgcn_permlane_bcast ||
8160 IID == Intrinsic::amdgcn_permlane_up ||
8161 IID == Intrinsic::amdgcn_permlane_down ||
8162 IID == Intrinsic::amdgcn_permlane_xor;
8163 SDLoc SL(N);
8164 MVT IntVT = MVT::getIntegerVT(ValSize);
8165 const GCNSubtarget *ST = TLI.getSubtarget();
8166
8167 unsigned SplitSize = 32;
8168 if (IID == Intrinsic::amdgcn_update_dpp && (ValSize % 64 == 0) &&
8169 ST->hasDPALU_DPP() &&
8170 AMDGPU::isLegalDPALU_DPPControl(*ST, N->getConstantOperandVal(3)))
8171 SplitSize = 64;
8172
8173 auto createLaneOp = [&DAG, &SL, N, IID](SDValue Src0, SDValue Src1,
8174 SDValue Src2, MVT ValT) -> SDValue {
8176 switch (IID) {
8177 case Intrinsic::amdgcn_permlane16:
8178 case Intrinsic::amdgcn_permlanex16:
8179 case Intrinsic::amdgcn_update_dpp:
8180 Operands.push_back(N->getOperand(6));
8181 Operands.push_back(N->getOperand(5));
8182 Operands.push_back(N->getOperand(4));
8183 [[fallthrough]];
8184 case Intrinsic::amdgcn_writelane:
8185 case Intrinsic::amdgcn_permlane_bcast:
8186 case Intrinsic::amdgcn_permlane_up:
8187 case Intrinsic::amdgcn_permlane_down:
8188 case Intrinsic::amdgcn_permlane_xor:
8189 Operands.push_back(Src2);
8190 [[fallthrough]];
8191 case Intrinsic::amdgcn_readlane:
8192 case Intrinsic::amdgcn_set_inactive:
8193 case Intrinsic::amdgcn_set_inactive_chain_arg:
8194 case Intrinsic::amdgcn_mov_dpp8:
8195 Operands.push_back(Src1);
8196 [[fallthrough]];
8197 case Intrinsic::amdgcn_readfirstlane:
8198 case Intrinsic::amdgcn_permlane64:
8199 Operands.push_back(Src0);
8200 break;
8201 default:
8202 llvm_unreachable("unhandled lane op");
8203 }
8204
8205 Operands.push_back(DAG.getTargetConstant(IID, SL, MVT::i32));
8206 std::reverse(Operands.begin(), Operands.end());
8207
8208 if (SDNode *GL = N->getGluedNode()) {
8209 assert(GL->getOpcode() == ISD::CONVERGENCECTRL_GLUE);
8210 GL = GL->getOperand(0).getNode();
8211 Operands.push_back(DAG.getNode(ISD::CONVERGENCECTRL_GLUE, SL, MVT::Glue,
8212 SDValue(GL, 0)));
8213 }
8214
8215 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, ValT, Operands);
8216 };
8217
8218 SDValue Src0 = N->getOperand(1);
8219 SDValue Src1, Src2;
8220 if (IID == Intrinsic::amdgcn_readlane || IID == Intrinsic::amdgcn_writelane ||
8221 IID == Intrinsic::amdgcn_mov_dpp8 ||
8222 IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16 ||
8223 IsPermlaneShuffle) {
8224 Src1 = N->getOperand(2);
8225 if (IID == Intrinsic::amdgcn_writelane ||
8226 IID == Intrinsic::amdgcn_update_dpp || IsPermLane16 ||
8227 IsPermlaneShuffle)
8228 Src2 = N->getOperand(3);
8229 }
8230
8231 if (ValSize == SplitSize) {
8232 // Already legal
8233 return SDValue();
8234 }
8235
8236 if (ValSize < 32) {
8237 bool IsFloat = VT.isFloatingPoint();
8238 Src0 = DAG.getAnyExtOrTrunc(IsFloat ? DAG.getBitcast(IntVT, Src0) : Src0,
8239 SL, MVT::i32);
8240
8241 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16) {
8242 Src1 = DAG.getAnyExtOrTrunc(IsFloat ? DAG.getBitcast(IntVT, Src1) : Src1,
8243 SL, MVT::i32);
8244 }
8245
8246 if (IID == Intrinsic::amdgcn_writelane) {
8247 Src2 = DAG.getAnyExtOrTrunc(IsFloat ? DAG.getBitcast(IntVT, Src2) : Src2,
8248 SL, MVT::i32);
8249 }
8250
8251 SDValue LaneOp = createLaneOp(Src0, Src1, Src2, MVT::i32);
8252 SDValue Trunc = DAG.getAnyExtOrTrunc(LaneOp, SL, IntVT);
8253 return IsFloat ? DAG.getBitcast(VT, Trunc) : Trunc;
8254 }
8255
8256 if (ValSize % SplitSize != 0)
8257 return SDValue();
8258
8259 auto unrollLaneOp = [&DAG, &SL](SDNode *N) -> SDValue {
8260 EVT VT = N->getValueType(0);
8261 unsigned NE = VT.getVectorNumElements();
8262 EVT EltVT = VT.getVectorElementType();
8264 unsigned NumOperands = N->getNumOperands();
8265 SmallVector<SDValue, 4> Operands(NumOperands);
8266 SDNode *GL = N->getGluedNode();
8267
8268 // only handle convergencectrl_glue
8270
8271 for (unsigned i = 0; i != NE; ++i) {
8272 for (unsigned j = 0, e = GL ? NumOperands - 1 : NumOperands; j != e;
8273 ++j) {
8274 SDValue Operand = N->getOperand(j);
8275 EVT OperandVT = Operand.getValueType();
8276 if (OperandVT.isVector()) {
8277 // A vector operand; extract a single element.
8278 EVT OperandEltVT = OperandVT.getVectorElementType();
8279 Operands[j] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, OperandEltVT,
8280 Operand, DAG.getVectorIdxConstant(i, SL));
8281 } else {
8282 // A scalar operand; just use it as is.
8283 Operands[j] = Operand;
8284 }
8285 }
8286
8287 if (GL)
8288 Operands[NumOperands - 1] =
8289 DAG.getNode(ISD::CONVERGENCECTRL_GLUE, SL, MVT::Glue,
8290 SDValue(GL->getOperand(0).getNode(), 0));
8291
8292 Scalars.push_back(DAG.getNode(N->getOpcode(), SL, EltVT, Operands));
8293 }
8294
8295 EVT VecVT = EVT::getVectorVT(*DAG.getContext(), EltVT, NE);
8296 return DAG.getBuildVector(VecVT, SL, Scalars);
8297 };
8298
8299 if (VT.isVector()) {
8300 switch (MVT::SimpleValueType EltTy =
8302 case MVT::i32:
8303 case MVT::f32:
8304 if (SplitSize == 32) {
8305 SDValue LaneOp = createLaneOp(Src0, Src1, Src2, VT.getSimpleVT());
8306 return unrollLaneOp(LaneOp.getNode());
8307 }
8308 [[fallthrough]];
8309 case MVT::i16:
8310 case MVT::f16:
8311 case MVT::bf16: {
8312 unsigned SubVecNumElt =
8313 SplitSize / VT.getVectorElementType().getSizeInBits();
8314 MVT SubVecVT = MVT::getVectorVT(EltTy, SubVecNumElt);
8316 SDValue Src0SubVec, Src1SubVec, Src2SubVec;
8317 for (unsigned i = 0, EltIdx = 0; i < ValSize / SplitSize; i++) {
8318 Src0SubVec = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, SubVecVT, Src0,
8319 DAG.getConstant(EltIdx, SL, MVT::i32));
8320
8321 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive ||
8322 IsPermLane16) {
8323 Src1SubVec = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, SubVecVT, Src1,
8324 DAG.getConstant(EltIdx, SL, MVT::i32));
8325
8326 Pieces.push_back(
8327 createLaneOp(Src0SubVec, Src1SubVec, Src2, SubVecVT));
8328 } else if (IID == Intrinsic::amdgcn_writelane) {
8329 Src2SubVec = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, SubVecVT, Src2,
8330 DAG.getConstant(EltIdx, SL, MVT::i32));
8331 Pieces.push_back(
8332 createLaneOp(Src0SubVec, Src1, Src2SubVec, SubVecVT));
8333 } else {
8334 Pieces.push_back(createLaneOp(Src0SubVec, Src1, Src2, SubVecVT));
8335 }
8336
8337 EltIdx += SubVecNumElt;
8338 }
8339 return DAG.getNode(ISD::CONCAT_VECTORS, SL, VT, Pieces);
8340 }
8341 default:
8342 // Handle all other cases by bitcasting to i32 vectors
8343 break;
8344 }
8345 }
8346
8347 MVT VecVT =
8348 MVT::getVectorVT(MVT::getIntegerVT(SplitSize), ValSize / SplitSize);
8349 Src0 = DAG.getBitcast(VecVT, Src0);
8350
8351 if (IID == Intrinsic::amdgcn_update_dpp || IsSetInactive || IsPermLane16)
8352 Src1 = DAG.getBitcast(VecVT, Src1);
8353
8354 if (IID == Intrinsic::amdgcn_writelane)
8355 Src2 = DAG.getBitcast(VecVT, Src2);
8356
8357 SDValue LaneOp = createLaneOp(Src0, Src1, Src2, VecVT);
8358 SDValue UnrolledLaneOp = unrollLaneOp(LaneOp.getNode());
8359 return DAG.getBitcast(VT, UnrolledLaneOp);
8360}
8361
8363 SelectionDAG &DAG) {
8364 EVT VT = N->getValueType(0);
8365
8366 if (VT.getSizeInBits() != 32)
8367 return SDValue();
8368
8369 SDLoc SL(N);
8370
8371 SDValue Value = N->getOperand(1);
8372 SDValue Index = N->getOperand(2);
8373
8374 // ds_bpermute requires index to be multiplied by 4
8375 SDValue ShiftAmount = DAG.getShiftAmountConstant(2, MVT::i32, SL);
8376 SDValue ShiftedIndex =
8377 DAG.getNode(ISD::SHL, SL, Index.getValueType(), Index, ShiftAmount);
8378
8379 // Intrinsics will require i32 to operate on
8380 SDValue ValueI32 = DAG.getBitcast(MVT::i32, Value);
8381
8382 auto MakeIntrinsic = [&DAG, &SL](unsigned IID, MVT RetVT,
8383 SmallVector<SDValue> IntrinArgs) -> SDValue {
8385 Operands[0] = DAG.getTargetConstant(IID, SL, MVT::i32);
8386 Operands.append(IntrinArgs);
8387 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, RetVT, Operands);
8388 };
8389
8390 // If we can bpermute across the whole wave, then just do that
8392 SDValue BPermute = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute, MVT::i32,
8393 {ShiftedIndex, ValueI32});
8394 return DAG.getBitcast(VT, BPermute);
8395 }
8396
8397 assert(TLI.getSubtarget()->isWave64());
8398
8399 // Otherwise, we need to make use of whole wave mode
8400 SDValue PoisonVal = DAG.getPOISON(ValueI32->getValueType(0));
8401
8402 // Set inactive lanes to poison
8403 SDValue WWMValue = MakeIntrinsic(Intrinsic::amdgcn_set_inactive, MVT::i32,
8404 {ValueI32, PoisonVal});
8405 SDValue WWMIndex = MakeIntrinsic(Intrinsic::amdgcn_set_inactive, MVT::i32,
8406 {ShiftedIndex, PoisonVal});
8407
8408 SDValue Swapped =
8409 MakeIntrinsic(Intrinsic::amdgcn_permlane64, MVT::i32, {WWMValue});
8410
8411 // Get permutation of each half, then we'll select which one to use
8412 SDValue BPermSameHalf = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute, MVT::i32,
8413 {WWMIndex, WWMValue});
8414 SDValue BPermOtherHalf = MakeIntrinsic(Intrinsic::amdgcn_ds_bpermute,
8415 MVT::i32, {WWMIndex, Swapped});
8416 SDValue BPermOtherHalfWWM =
8417 MakeIntrinsic(Intrinsic::amdgcn_wwm, MVT::i32, {BPermOtherHalf});
8418
8419 // Select which side to take the permute from
8420 SDValue ThreadIDMask = DAG.getAllOnesConstant(SL, MVT::i32);
8421 // We can get away with only using mbcnt_lo here since we're only
8422 // trying to detect which side of 32 each lane is on, and mbcnt_lo
8423 // returns 32 for lanes 32-63.
8424 SDValue ThreadID =
8425 MakeIntrinsic(Intrinsic::amdgcn_mbcnt_lo, MVT::i32,
8426 {ThreadIDMask, DAG.getTargetConstant(0, SL, MVT::i32)});
8427
8428 SDValue SameOrOtherHalf =
8429 DAG.getNode(ISD::AND, SL, MVT::i32,
8430 DAG.getNode(ISD::XOR, SL, MVT::i32, ThreadID, Index),
8431 DAG.getTargetConstant(32, SL, MVT::i32));
8432 SDValue UseSameHalf =
8433 DAG.getSetCC(SL, MVT::i1, SameOrOtherHalf,
8434 DAG.getConstant(0, SL, MVT::i32), ISD::SETEQ);
8435 SDValue Result = DAG.getSelect(SL, MVT::i32, UseSameHalf, BPermSameHalf,
8436 BPermOtherHalfWWM);
8437 return DAG.getBitcast(VT, Result);
8438}
8439
8442 SelectionDAG &DAG) const {
8443 switch (N->getOpcode()) {
8445 if (SDValue Res = lowerINSERT_VECTOR_ELT(SDValue(N, 0), DAG))
8446 Results.push_back(Res);
8447 return;
8448 }
8450 if (SDValue Res = lowerEXTRACT_VECTOR_ELT(SDValue(N, 0), DAG))
8451 Results.push_back(Res);
8452 return;
8453 }
8455 if (SDValue Res = LowerCONVERT_TO_ARBITRARY_FP(SDValue(N, 0), DAG))
8456 Results.push_back(Res);
8457 return;
8458 }
8460 unsigned IID = N->getConstantOperandVal(0);
8461 switch (IID) {
8462 case Intrinsic::amdgcn_wave_reduce_min:
8463 case Intrinsic::amdgcn_wave_reduce_umin:
8464 case Intrinsic::amdgcn_wave_reduce_max:
8465 case Intrinsic::amdgcn_wave_reduce_umax:
8466 case Intrinsic::amdgcn_wave_reduce_add:
8467 case Intrinsic::amdgcn_wave_reduce_sub:
8468 case Intrinsic::amdgcn_wave_reduce_and:
8469 case Intrinsic::amdgcn_wave_reduce_or:
8470 case Intrinsic::amdgcn_wave_reduce_xor: {
8471 EVT VT = N->getValueType(0);
8472 if (isTypeLegal(VT))
8473 return;
8474 SDLoc SL(N);
8475 bool NeedsSignExt = IID == Intrinsic::amdgcn_wave_reduce_min ||
8476 IID == Intrinsic::amdgcn_wave_reduce_max ||
8477 IID == Intrinsic::amdgcn_wave_reduce_add ||
8478 IID == Intrinsic::amdgcn_wave_reduce_sub;
8479 unsigned ExtOpc = NeedsSignExt ? ISD::SIGN_EXTEND : ISD::ZERO_EXTEND;
8480 SDValue ExtSrc = DAG.getNode(ExtOpc, SL, MVT::i32, N->getOperand(1));
8481 SDValue Result = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
8482 N->getOperand(0), ExtSrc, N->getOperand(2));
8483 Results.push_back(DAG.getNode(ISD::TRUNCATE, SL, VT, Result));
8484 return;
8485 }
8486 case Intrinsic::amdgcn_make_buffer_rsrc:
8487 Results.push_back(lowerPointerAsRsrcIntrin(N, DAG));
8488 return;
8489 case Intrinsic::amdgcn_cvt_pkrtz: {
8490 SDValue Src0 = N->getOperand(1);
8491 SDValue Src1 = N->getOperand(2);
8492 SDLoc SL(N);
8493 SDValue Cvt =
8494 DAG.getNode(AMDGPUISD::CVT_PKRTZ_F16_F32, SL, MVT::i32, Src0, Src1);
8495 Results.push_back(DAG.getNode(ISD::BITCAST, SL, MVT::v2f16, Cvt));
8496 return;
8497 }
8498 case Intrinsic::amdgcn_cvt_pknorm_i16:
8499 case Intrinsic::amdgcn_cvt_pknorm_u16:
8500 case Intrinsic::amdgcn_cvt_pk_i16:
8501 case Intrinsic::amdgcn_cvt_pk_u16: {
8502 SDValue Src0 = N->getOperand(1);
8503 SDValue Src1 = N->getOperand(2);
8504 SDLoc SL(N);
8505 unsigned Opcode;
8506
8507 if (IID == Intrinsic::amdgcn_cvt_pknorm_i16)
8508 Opcode = AMDGPUISD::CVT_PKNORM_I16_F32;
8509 else if (IID == Intrinsic::amdgcn_cvt_pknorm_u16)
8510 Opcode = AMDGPUISD::CVT_PKNORM_U16_F32;
8511 else if (IID == Intrinsic::amdgcn_cvt_pk_i16)
8512 Opcode = AMDGPUISD::CVT_PK_I16_I32;
8513 else
8514 Opcode = AMDGPUISD::CVT_PK_U16_U32;
8515
8516 EVT VT = N->getValueType(0);
8517 if (isTypeLegal(VT))
8518 Results.push_back(DAG.getNode(Opcode, SL, VT, Src0, Src1));
8519 else {
8520 SDValue Cvt = DAG.getNode(Opcode, SL, MVT::i32, Src0, Src1);
8521 Results.push_back(DAG.getNode(ISD::BITCAST, SL, MVT::v2i16, Cvt));
8522 }
8523 return;
8524 }
8525 case Intrinsic::amdgcn_s_buffer_load: {
8526 SDValue Op = SDValue(N, 0);
8527 EVT VT = Op.getValueType();
8528 Results.push_back(lowerSBuffer(VT, VT, SDLoc(Op), DAG.getEntryNode(),
8529 Op.getOperand(1), Op.getOperand(2),
8530 Op.getOperand(3), DAG));
8531 return;
8532 }
8533 case Intrinsic::amdgcn_dead: {
8534 for (unsigned I = 0, E = N->getNumValues(); I < E; ++I)
8535 Results.push_back(DAG.getPOISON(N->getValueType(I)));
8536 return;
8537 }
8538 }
8539 break;
8540 }
8542 if (N->getConstantOperandVal(1) != Intrinsic::amdgcn_ptr_s_buffer_load &&
8543 N->getValueType(0).isSimple() &&
8544 SBufferLoadDiagnosticVTs[N->getSimpleValueType(0).SimpleTy])
8545 break;
8546 if (SDValue Res = LowerINTRINSIC_W_CHAIN(SDValue(N, 0), DAG)) {
8547 if (Res.getOpcode() == ISD::MERGE_VALUES) {
8548 // FIXME: Hacky
8549 for (unsigned I = 0; I < Res.getNumOperands(); I++) {
8550 Results.push_back(Res.getOperand(I));
8551 }
8552 } else {
8553 for (unsigned I = 0; I < N->getNumValues(); ++I)
8554 Results.push_back(Res.getValue(I));
8555 }
8556 return;
8557 }
8558
8559 break;
8560 }
8561 case ISD::SELECT: {
8562 SDLoc SL(N);
8563 EVT VT = N->getValueType(0);
8564 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VT);
8565 SDValue LHS = DAG.getNode(ISD::BITCAST, SL, NewVT, N->getOperand(1));
8566 SDValue RHS = DAG.getNode(ISD::BITCAST, SL, NewVT, N->getOperand(2));
8567
8568 EVT SelectVT = NewVT;
8569 if (NewVT.bitsLT(MVT::i32)) {
8570 LHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, LHS);
8571 RHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, RHS);
8572 SelectVT = MVT::i32;
8573 }
8574
8575 SDValue NewSelect =
8576 DAG.getNode(ISD::SELECT, SL, SelectVT, N->getOperand(0), LHS, RHS);
8577
8578 if (NewVT != SelectVT)
8579 NewSelect = DAG.getNode(ISD::TRUNCATE, SL, NewVT, NewSelect);
8580 Results.push_back(DAG.getNode(ISD::BITCAST, SL, VT, NewSelect));
8581 return;
8582 }
8583 case ISD::FNEG: {
8584 if (N->getValueType(0) != MVT::v2f16)
8585 break;
8586
8587 SDLoc SL(N);
8588 SDValue BC = DAG.getNode(ISD::BITCAST, SL, MVT::i32, N->getOperand(0));
8589
8590 SDValue Op = DAG.getNode(ISD::XOR, SL, MVT::i32, BC,
8591 DAG.getConstant(0x80008000, SL, MVT::i32));
8592 Results.push_back(DAG.getNode(ISD::BITCAST, SL, MVT::v2f16, Op));
8593 return;
8594 }
8595 case ISD::FABS: {
8596 if (N->getValueType(0) != MVT::v2f16)
8597 break;
8598
8599 SDLoc SL(N);
8600 SDValue BC = DAG.getNode(ISD::BITCAST, SL, MVT::i32, N->getOperand(0));
8601
8602 SDValue Op = DAG.getNode(ISD::AND, SL, MVT::i32, BC,
8603 DAG.getConstant(0x7fff7fff, SL, MVT::i32));
8604 Results.push_back(DAG.getNode(ISD::BITCAST, SL, MVT::v2f16, Op));
8605 return;
8606 }
8607 case ISD::FSQRT: {
8608 if (N->getValueType(0) != MVT::f16)
8609 break;
8610 Results.push_back(lowerFSQRTF16(SDValue(N, 0), DAG));
8611 break;
8612 }
8613 default:
8615 break;
8616 }
8617}
8618
8619/// Helper function for LowerBRCOND
8620static SDNode *findUser(SDValue Value, unsigned Opcode) {
8621
8622 for (SDUse &U : Value->uses()) {
8623 if (U.get() != Value)
8624 continue;
8625
8626 if (U.getUser()->getOpcode() == Opcode)
8627 return U.getUser();
8628 }
8629 return nullptr;
8630}
8631
8632unsigned SITargetLowering::isCFIntrinsic(const SDNode *Intr) const {
8633 if (Intr->getOpcode() == ISD::INTRINSIC_W_CHAIN) {
8634 switch (Intr->getConstantOperandVal(1)) {
8635 case Intrinsic::amdgcn_if:
8636 return AMDGPUISD::IF;
8637 case Intrinsic::amdgcn_else:
8638 return AMDGPUISD::ELSE;
8639 case Intrinsic::amdgcn_loop:
8640 return AMDGPUISD::LOOP;
8641 case Intrinsic::amdgcn_end_cf:
8642 llvm_unreachable("should not occur");
8643 default:
8644 return 0;
8645 }
8646 }
8647
8648 // break, if_break, else_break are all only used as inputs to loop, not
8649 // directly as branch conditions.
8650 return 0;
8651}
8652
8659
8661 if (Subtarget->isAmdPalOS() || Subtarget->isMesa3DOS())
8662 return false;
8663
8664 // FIXME: Either avoid relying on address space here or change the default
8665 // address space for functions to avoid the explicit check.
8666 return (GV->getValueType()->isFunctionTy() ||
8669}
8670
8672 return !shouldEmitFixup(GV) && !shouldEmitGOTReloc(GV);
8673}
8674
8676 if (!GV->hasExternalLinkage())
8677 return true;
8678
8679 // With object linking, external LDS declarations need relocations so the
8680 // linker can assign their offsets.
8682 if (const auto *GVar = dyn_cast<GlobalVariable>(GV)) {
8683 if (GVar->getAddressSpace() == AMDGPUAS::LOCAL_ADDRESS ||
8684 GVar->getAddressSpace() == AMDGPUAS::BARRIER) {
8685 assert(GVar->isDeclaration() &&
8686 "AS 3 & 13 GVs should be declaration here "
8687 "when object linking is enabled");
8688 return false;
8689 }
8690 }
8691 }
8692
8693 const auto OS = getTargetMachine().getTargetTriple().getOS();
8694 return OS == Triple::AMDHSA || OS == Triple::AMDPAL;
8695}
8696
8697/// This transforms the control flow intrinsics to get the branch destination as
8698/// last parameter, also switches branch target with BR if the need arise
8699SDValue SITargetLowering::LowerBRCOND(SDValue BRCOND, SelectionDAG &DAG) const {
8700 SDLoc DL(BRCOND);
8701
8702 SDNode *Intr = BRCOND.getOperand(1).getNode();
8703 SDValue Target = BRCOND.getOperand(2);
8704 SDNode *BR = nullptr;
8705 SDNode *SetCC = nullptr;
8706
8707 switch (Intr->getOpcode()) {
8708 case ISD::SETCC: {
8709 // As long as we negate the condition everything is fine
8710 SetCC = Intr;
8711 Intr = SetCC->getOperand(0).getNode();
8712 break;
8713 }
8714 case ISD::XOR: {
8715 // Similar to SETCC, if we have (xor c, -1), we will be fine.
8716 SDValue LHS = Intr->getOperand(0);
8717 SDValue RHS = Intr->getOperand(1);
8718 if (auto *C = dyn_cast<ConstantSDNode>(RHS); C && C->getZExtValue()) {
8719 Intr = LHS.getNode();
8720 break;
8721 }
8722 [[fallthrough]];
8723 }
8724 default: {
8725 // Get the target from BR if we don't negate the condition
8726 BR = findUser(BRCOND, ISD::BR);
8727 assert(BR && "brcond missing unconditional branch user");
8728 Target = BR->getOperand(1);
8729 }
8730 }
8731
8732 unsigned CFNode = isCFIntrinsic(Intr);
8733 if (CFNode == 0) {
8734 // This is a uniform branch so we don't need to legalize.
8735 return BRCOND;
8736 }
8737
8738 bool HaveChain = Intr->getOpcode() == ISD::INTRINSIC_VOID ||
8740
8741 assert(!SetCC ||
8742 (SetCC->getConstantOperandVal(1) == 1 &&
8743 cast<CondCodeSDNode>(SetCC->getOperand(2).getNode())->get() ==
8744 ISD::SETNE));
8745
8746 // operands of the new intrinsic call
8748 if (HaveChain)
8749 Ops.push_back(BRCOND.getOperand(0));
8750
8751 Ops.append(Intr->op_begin() + (HaveChain ? 2 : 1), Intr->op_end());
8752 Ops.push_back(Target);
8753
8754 ArrayRef<EVT> Res(Intr->value_begin() + 1, Intr->value_end());
8755
8756 // build the new intrinsic call
8757 SDNode *Result = DAG.getNode(CFNode, DL, DAG.getVTList(Res), Ops).getNode();
8758
8759 if (!HaveChain) {
8760 SDValue Ops[] = {SDValue(Result, 0), BRCOND.getOperand(0)};
8761
8763 }
8764
8765 if (BR) {
8766 // Give the branch instruction our target
8767 SDValue Ops[] = {BR->getOperand(0), BRCOND.getOperand(2)};
8768 SDValue NewBR = DAG.getNode(ISD::BR, DL, BR->getVTList(), Ops);
8769 DAG.ReplaceAllUsesWith(BR, NewBR.getNode());
8770 }
8771
8772 SDValue Chain = SDValue(Result, Result->getNumValues() - 1);
8773
8774 // Copy the intrinsic results to registers
8775 for (unsigned i = 1, e = Intr->getNumValues() - 1; i != e; ++i) {
8776 SDNode *CopyToReg = findUser(SDValue(Intr, i), ISD::CopyToReg);
8777 if (!CopyToReg)
8778 continue;
8779
8780 Chain = DAG.getCopyToReg(Chain, DL, CopyToReg->getOperand(1),
8781 SDValue(Result, i - 1), SDValue());
8782
8783 DAG.ReplaceAllUsesWith(SDValue(CopyToReg, 0), CopyToReg->getOperand(0));
8784 }
8785
8786 // Remove the old intrinsic from the chain
8787 DAG.ReplaceAllUsesOfValueWith(SDValue(Intr, Intr->getNumValues() - 1),
8788 Intr->getOperand(0));
8789
8790 return Chain;
8791}
8792
8793SDValue SITargetLowering::LowerRETURNADDR(SDValue Op, SelectionDAG &DAG) const {
8794 MVT VT = Op.getSimpleValueType();
8795 SDLoc DL(Op);
8796 // Checking the depth
8797 if (Op.getConstantOperandVal(0) != 0)
8798 return DAG.getConstant(0, DL, VT);
8799
8801 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
8802 // Check for kernel and shader functions
8803 if (Info->isEntryFunction())
8804 return DAG.getConstant(0, DL, VT);
8805
8806 MachineFrameInfo &MFI = MF.getFrameInfo();
8807 // There is a call to @llvm.returnaddress in this function
8808 MFI.setReturnAddressIsTaken(true);
8809
8810 const SIRegisterInfo *TRI = getSubtarget()->getRegisterInfo();
8811 // Get the return address reg and mark it as an implicit live-in
8812 Register Reg = MF.addLiveIn(TRI->getReturnAddressReg(MF),
8813 getRegClassFor(VT, Op.getNode()->isDivergent()));
8814
8815 return DAG.getCopyFromReg(DAG.getEntryNode(), DL, Reg, VT);
8816}
8817
8818SDValue SITargetLowering::LowerSPONENTRY(SDValue Op, SelectionDAG &DAG) const {
8820 SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
8821
8822 // For functions that set up their own stack, select the GET_STACK_BASE
8823 // pseudo.
8824 if (MFI->isBottomOfStack())
8825 return Op;
8826
8827 // For everything else, create a dummy stack object.
8828 int FI = MF.getFrameInfo().CreateFixedObject(1, 0, /*IsImmutable=*/false);
8829 return DAG.getFrameIndex(FI, Op.getValueType());
8830}
8831
8832SDValue SITargetLowering::getFPExtOrFPRound(SelectionDAG &DAG, SDValue Op,
8833 const SDLoc &DL, EVT VT) const {
8834 return Op.getValueType().bitsLE(VT)
8835 ? DAG.getNode(ISD::FP_EXTEND, DL, VT, Op)
8836 : DAG.getNode(ISD::FP_ROUND, DL, VT, Op,
8837 DAG.getTargetConstant(0, DL, MVT::i32));
8838}
8839
8840SDValue SITargetLowering::splitFP_ROUNDVectorOp(SDValue Op,
8841 SelectionDAG &DAG) const {
8842 EVT DstVT = Op.getValueType();
8843 unsigned NumElts = DstVT.getVectorNumElements();
8844 assert(NumElts > 2 && isPowerOf2_32(NumElts));
8845
8846 auto [Lo, Hi] = DAG.SplitVectorOperand(Op.getNode(), 0);
8847
8848 SDLoc DL(Op);
8849 unsigned Opc = Op.getOpcode();
8850 SDValue Flags = Op.getOperand(1);
8851 EVT HalfDstVT =
8852 EVT::getVectorVT(*DAG.getContext(), DstVT.getScalarType(), NumElts / 2);
8853 SDValue OpLo = DAG.getNode(Opc, DL, HalfDstVT, Lo, Flags);
8854 SDValue OpHi = DAG.getNode(Opc, DL, HalfDstVT, Hi, Flags);
8855
8856 return DAG.getNode(ISD::CONCAT_VECTORS, DL, DstVT, OpLo, OpHi);
8857}
8858
8859SDValue SITargetLowering::lowerFP_ROUND(SDValue Op, SelectionDAG &DAG) const {
8860 bool IsStrict = Op->isStrictFPOpcode();
8861 SDValue Src = Op.getOperand(IsStrict ? 1 : 0);
8862 EVT SrcVT = Src.getValueType();
8863 EVT DstVT = Op.getValueType();
8864
8865 if (DstVT.isVectorOf(MVT::f16)) {
8866 assert(Subtarget->hasCvtPkF16F32Inst() && "support v_cvt_pk_f16_f32");
8867 if (SrcVT.getScalarType() != MVT::f32)
8868 return SDValue();
8869 return SrcVT == MVT::v2f32 ? Op : splitFP_ROUNDVectorOp(Op, DAG);
8870 }
8871
8872 if (SrcVT.getScalarType() != MVT::f64)
8873 return Op;
8874
8875 SDLoc DL(Op);
8876 if (DstVT == MVT::f16) {
8877 // TODO: Handle strictfp
8878 if (Op.getOpcode() != ISD::FP_ROUND)
8879 return Op;
8880
8881 if (!Subtarget->has16BitInsts()) {
8882 SDValue FpToFp16 = DAG.getNode(ISD::FP_TO_FP16, DL, MVT::i32, Src);
8883 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToFp16);
8884 return DAG.getNode(ISD::BITCAST, DL, MVT::f16, Trunc);
8885 }
8886 if (Op->getFlags().hasApproximateFuncs()) {
8887 SDValue Flags = Op.getOperand(1);
8888 SDValue Src32 = DAG.getNode(ISD::FP_ROUND, DL, MVT::f32, Src, Flags);
8889 return DAG.getNode(ISD::FP_ROUND, DL, MVT::f16, Src32, Flags);
8890 }
8891 SDValue FpToFp16 = LowerF64ToF16Safe(Src, DL, DAG);
8892 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, FpToFp16);
8893 return DAG.getNode(ISD::BITCAST, DL, MVT::f16, Trunc);
8894 }
8895
8896 assert(DstVT.getScalarType() == MVT::bf16 &&
8897 "custom lower FP_ROUND for f16 or bf16");
8898 assert(Subtarget->hasBF16ConversionInsts() && "f32 -> bf16 is legal");
8899
8900 // Round-inexact-to-odd f64 to f32, then do the final rounding using the
8901 // hardware f32 -> bf16 instruction.
8902 EVT F32VT = SrcVT.changeElementType(*DAG.getContext(), MVT::f32);
8903 SDValue Rod = expandRoundInexactToOdd(F32VT, Src, DL, DAG);
8904 if (IsStrict) {
8905 return DAG.getNode(
8906 ISD::STRICT_FP_ROUND, DL, {DstVT, MVT::Other},
8907 {Op.getOperand(0), Rod, DAG.getTargetConstant(0, DL, MVT::i32)});
8908 }
8909 return DAG.getNode(ISD::FP_ROUND, DL, DstVT, Rod,
8910 DAG.getTargetConstant(0, DL, MVT::i32));
8911}
8912
8913SDValue SITargetLowering::lowerFMINNUM_FMAXNUM(SDValue Op,
8914 SelectionDAG &DAG) const {
8915 EVT VT = Op.getValueType();
8916 const MachineFunction &MF = DAG.getMachineFunction();
8917 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
8918 bool IsIEEEMode = Info->getMode().IEEE;
8919
8920 // FIXME: Assert during selection that this is only selected for
8921 // ieee_mode. Currently a combine can produce the ieee version for non-ieee
8922 // mode functions, but this happens to be OK since it's only done in cases
8923 // where there is known no sNaN.
8924 if (IsIEEEMode && !Subtarget->hasIEEEMinimumMaximumInsts())
8925 return expandFMINNUM_FMAXNUM(Op.getNode(), DAG);
8926
8927 if (VT == MVT::v4f16 || VT == MVT::v8f16 || VT == MVT::v16f16 ||
8928 VT == MVT::v32f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16 ||
8929 VT == MVT::v16bf16 || VT == MVT::v32bf16 || VT == MVT::v4f64 ||
8930 VT == MVT::v8f64 || VT == MVT::v16f64 || VT == MVT::v32f64)
8931 return splitBinaryVectorOp(Op, DAG);
8932 return Op;
8933}
8934
8935SDValue
8936SITargetLowering::lowerFMINIMUMNUM_FMAXIMUMNUM(SDValue Op,
8937 SelectionDAG &DAG) const {
8938 EVT VT = Op.getValueType();
8939 const MachineFunction &MF = DAG.getMachineFunction();
8940 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
8941 bool IsIEEEMode = Info->getMode().IEEE;
8942
8943 if (IsIEEEMode && !Subtarget->hasIEEEMinimumMaximumInsts())
8944 return expandFMINIMUMNUM_FMAXIMUMNUM(Op.getNode(), DAG);
8945
8946 if (VT == MVT::v4f16 || VT == MVT::v8f16 || VT == MVT::v16f16 ||
8947 VT == MVT::v32f16 || VT == MVT::v4bf16 || VT == MVT::v8bf16 ||
8948 VT == MVT::v16bf16 || VT == MVT::v32bf16 || VT == MVT::v4f64 ||
8949 VT == MVT::v8f64 || VT == MVT::v16f64 || VT == MVT::v32f64)
8950 return splitBinaryVectorOp(Op, DAG);
8951 return Op;
8952}
8953
8954SDValue SITargetLowering::lowerFLDEXP(SDValue Op, SelectionDAG &DAG) const {
8955 bool IsStrict = Op.getOpcode() == ISD::STRICT_FLDEXP;
8956 EVT VT = Op.getValueType();
8957 assert(VT == MVT::f16);
8958
8959 SDValue Exp = Op.getOperand(IsStrict ? 2 : 1);
8960 EVT ExpVT = Exp.getValueType();
8961 if (ExpVT == MVT::i16)
8962 return Op;
8963
8964 SDLoc DL(Op);
8965
8966 // Correct the exponent type for f16 to i16.
8967 // Clamp the range of the exponent to the instruction's range.
8968
8969 // TODO: This should be a generic narrowing legalization, and can easily be
8970 // for GlobalISel.
8971
8972 SDValue MinExp = DAG.getSignedConstant(minIntN(16), DL, ExpVT);
8973 SDValue ClampMin = DAG.getNode(ISD::SMAX, DL, ExpVT, Exp, MinExp);
8974
8975 SDValue MaxExp = DAG.getSignedConstant(maxIntN(16), DL, ExpVT);
8976 SDValue Clamp = DAG.getNode(ISD::SMIN, DL, ExpVT, ClampMin, MaxExp);
8977
8978 SDValue TruncExp = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, Clamp);
8979
8980 if (IsStrict) {
8981 return DAG.getNode(ISD::STRICT_FLDEXP, DL, {VT, MVT::Other},
8982 {Op.getOperand(0), Op.getOperand(1), TruncExp});
8983 }
8984
8985 return DAG.getNode(ISD::FLDEXP, DL, VT, Op.getOperand(0), TruncExp);
8986}
8987
8989 switch (Op->getOpcode()) {
8990 case ISD::ABS:
8991 case ISD::SRA:
8992 case ISD::SMIN:
8993 case ISD::SMAX:
8994 return ISD::SIGN_EXTEND;
8995 case ISD::SRL:
8996 case ISD::UMIN:
8997 case ISD::UMAX:
8998 case ISD::USUBSAT:
8999 case ISD::UADDSAT:
9000 return ISD::ZERO_EXTEND;
9001 case ISD::ADD:
9002 case ISD::SUB:
9003 case ISD::AND:
9004 case ISD::OR:
9005 case ISD::XOR:
9006 case ISD::SHL:
9007 case ISD::SELECT:
9008 case ISD::MUL:
9009 // operation result won't be influenced by garbage high bits.
9010 // TODO: are all of those cases correct, and are there more?
9011 return ISD::ANY_EXTEND;
9012 case ISD::SETCC: {
9013 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(2))->get();
9015 }
9016 default:
9017 llvm_unreachable("unexpected opcode!");
9018 }
9019}
9020
9021SDValue
9022SITargetLowering::promoteUniformUnaryOpToI32(SDValue Op,
9023 DAGCombinerInfo &DCI) const {
9024 EVT OpTy = Op.getValueType();
9025 SelectionDAG &DAG = DCI.DAG;
9026 EVT ExtTy = OpTy.changeElementType(*DAG.getContext(), MVT::i32);
9027
9028 if (isNarrowingProfitable(Op.getNode(), ExtTy, OpTy))
9029 return SDValue();
9030
9031 SDLoc DL(Op);
9032 SDValue Input = Op.getOperand(0);
9033 const unsigned ExtOp = getExtOpcodeForPromotedOp(Op);
9034 Input = DAG.getNode(ExtOp, DL, ExtTy, Input);
9035
9036 SDValue NewVal = DAG.getNode(Op.getOpcode(), DL, ExtTy, Input);
9037
9038 return DAG.getNode(ISD::TRUNCATE, DL, OpTy, NewVal);
9039}
9040
9041SDValue SITargetLowering::promoteUniformOpToI32(SDValue Op,
9042 DAGCombinerInfo &DCI) const {
9043 const unsigned Opc = Op.getOpcode();
9044 assert(Opc == ISD::ADD || Opc == ISD::SUB || Opc == ISD::SHL ||
9045 Opc == ISD::SRL || Opc == ISD::SRA || Opc == ISD::AND ||
9046 Opc == ISD::OR || Opc == ISD::XOR || Opc == ISD::MUL ||
9047 Opc == ISD::SETCC || Opc == ISD::SELECT || Opc == ISD::SMIN ||
9048 Opc == ISD::SMAX || Opc == ISD::UMIN || Opc == ISD::UMAX ||
9049 Opc == ISD::USUBSAT || Opc == ISD::UADDSAT);
9050
9051 EVT OpTy = (Opc != ISD::SETCC) ? Op.getValueType()
9052 : Op->getOperand(0).getValueType();
9053 auto &DAG = DCI.DAG;
9054 auto ExtTy = OpTy.changeElementType(*DAG.getContext(), MVT::i32);
9055
9056 if (DCI.isBeforeLegalizeOps() ||
9057 isNarrowingProfitable(Op.getNode(), ExtTy, OpTy))
9058 return SDValue();
9059
9060 SDLoc DL(Op);
9061 SDValue LHS;
9062 SDValue RHS;
9063 if (Opc == ISD::SELECT) {
9064 LHS = Op->getOperand(1);
9065 RHS = Op->getOperand(2);
9066 } else {
9067 LHS = Op->getOperand(0);
9068 RHS = Op->getOperand(1);
9069 }
9070
9071 const unsigned ExtOp = getExtOpcodeForPromotedOp(Op);
9072 LHS = DAG.getNode(ExtOp, DL, ExtTy, {LHS});
9073
9074 // Special case: for shifts, the RHS always needs a zext.
9075 if (Opc == ISD::SHL || Opc == ISD::SRL || Opc == ISD::SRA)
9076 RHS = DAG.getNode(ISD::ZERO_EXTEND, DL, ExtTy, {RHS});
9077 else
9078 RHS = DAG.getNode(ExtOp, DL, ExtTy, {RHS});
9079
9080 // setcc always return i1/i1 vec so no need to truncate after.
9081 if (Opc == ISD::SETCC) {
9082 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(2))->get();
9083 return DAG.getSetCC(DL, Op.getValueType(), LHS, RHS, CC);
9084 }
9085
9086 // For other ops, we extend the operation's return type as well so we need to
9087 // truncate back to the original type.
9088 SDValue NewVal;
9089 if (Opc == ISD::SELECT)
9090 NewVal = DAG.getNode(ISD::SELECT, DL, ExtTy, {Op->getOperand(0), LHS, RHS});
9091 else if (Opc == ISD::UADDSAT) {
9092 SDValue Sum = DAG.getNode(ISD::ADD, DL, ExtTy, LHS, RHS);
9093 SDValue MaxVal = DAG.getConstant(
9094 APInt::getMaxValue(OpTy.getScalarSizeInBits()).zext(32), DL, ExtTy);
9095 NewVal = DAG.getNode(ISD::UMIN, DL, ExtTy, Sum, MaxVal);
9096 } else
9097 NewVal = DAG.getNode(Opc, DL, ExtTy, {LHS, RHS});
9098
9099 return DAG.getZExtOrTrunc(NewVal, DL, OpTy);
9100}
9101
9102SDValue SITargetLowering::lowerFCOPYSIGN(SDValue Op, SelectionDAG &DAG) const {
9103 SDValue Mag = Op.getOperand(0);
9104 EVT MagVT = Mag.getValueType();
9105
9106 if (MagVT.getVectorNumElements() > 2)
9107 return splitBinaryVectorOp(Op, DAG);
9108
9109 SDValue Sign = Op.getOperand(1);
9110 EVT SignVT = Sign.getValueType();
9111
9112 if (MagVT == SignVT)
9113 return Op;
9114
9115 // fcopysign v2f16:mag, v2f32:sign ->
9116 // fcopysign v2f16:mag,
9117 // bitcast (trunc (srl (bitcast sign to v2i32), 16) to v2i16)
9118
9119 SDLoc SL(Op);
9120 SDValue SignAsInt32 = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Sign);
9121 SDValue ShiftAmt = DAG.getShiftAmountConstant(16, MVT::v2i32, SL);
9122 SDValue SignShifted =
9123 DAG.getNode(ISD::SRL, SL, MVT::v2i32, SignAsInt32, ShiftAmt);
9124 SDValue SignAsInt16 = DAG.getNode(ISD::TRUNCATE, SL, MVT::v2i16, SignShifted);
9125
9126 SDValue SignAsHalf16 = DAG.getNode(ISD::BITCAST, SL, MagVT, SignAsInt16);
9127
9128 return DAG.getNode(ISD::FCOPYSIGN, SL, MagVT, Mag, SignAsHalf16);
9129}
9130
9131// Custom lowering for vector multiplications and s_mul_u64.
9132SDValue SITargetLowering::lowerMUL(SDValue Op, SelectionDAG &DAG) const {
9133 EVT VT = Op.getValueType();
9134
9135 // Split vector operands.
9136 if (VT.isVector())
9137 return splitBinaryVectorOp(Op, DAG);
9138
9139 assert(VT == MVT::i64 && "The following code is a special for s_mul_u64");
9140
9141 // There are four ways to lower s_mul_u64:
9142 //
9143 // 1. If all the operands are uniform, then we lower it as it is.
9144 //
9145 // 2. If the operands are divergent, then we have to split s_mul_u64 in 32-bit
9146 // multiplications because there is not a vector equivalent of s_mul_u64.
9147 //
9148 // 3. If the cost model decides that it is more efficient to use vector
9149 // registers, then we have to split s_mul_u64 in 32-bit multiplications.
9150 // This happens in splitScalarSMULU64() in SIInstrInfo.cpp .
9151 //
9152 // 4. If the cost model decides to use vector registers and both of the
9153 // operands are zero-extended/sign-extended from 32-bits, then we split the
9154 // s_mul_u64 in two 32-bit multiplications. The problem is that it is not
9155 // possible to check if the operands are zero-extended or sign-extended in
9156 // SIInstrInfo.cpp. For this reason, here, we replace s_mul_u64 with
9157 // s_mul_u64_u32_pseudo if both operands are zero-extended and we replace
9158 // s_mul_u64 with s_mul_i64_i32_pseudo if both operands are sign-extended.
9159 // If the cost model decides that we have to use vector registers, then
9160 // splitScalarSMulPseudo() (in SIInstrInfo.cpp) split s_mul_u64_u32/
9161 // s_mul_i64_i32_pseudo in two vector multiplications. If the cost model
9162 // decides that we should use scalar registers, then s_mul_u64_u32_pseudo/
9163 // s_mul_i64_i32_pseudo is lowered as s_mul_u64 in expandPostRAPseudo() in
9164 // SIInstrInfo.cpp .
9165
9166 if (Op->isDivergent())
9167 return SDValue();
9168
9169 SDValue Op0 = Op.getOperand(0);
9170 SDValue Op1 = Op.getOperand(1);
9171 // If all the operands are zero-enteted to 32-bits, then we replace s_mul_u64
9172 // with s_mul_u64_u32_pseudo. If all the operands are sign-extended to
9173 // 32-bits, then we replace s_mul_u64 with s_mul_i64_i32_pseudo.
9174 KnownBits Op0KnownBits = DAG.computeKnownBits(Op0);
9175 unsigned Op0LeadingZeros = Op0KnownBits.countMinLeadingZeros();
9176 KnownBits Op1KnownBits = DAG.computeKnownBits(Op1);
9177 unsigned Op1LeadingZeros = Op1KnownBits.countMinLeadingZeros();
9178 SDLoc SL(Op);
9179 if (Op0LeadingZeros >= 32 && Op1LeadingZeros >= 32)
9180 return SDValue(
9181 DAG.getMachineNode(AMDGPU::S_MUL_U64_U32_PSEUDO, SL, VT, Op0, Op1), 0);
9182 unsigned Op0SignBits = DAG.ComputeNumSignBits(Op0);
9183 unsigned Op1SignBits = DAG.ComputeNumSignBits(Op1);
9184 if (Op0SignBits >= 33 && Op1SignBits >= 33)
9185 return SDValue(
9186 DAG.getMachineNode(AMDGPU::S_MUL_I64_I32_PSEUDO, SL, VT, Op0, Op1), 0);
9187 // If all the operands are uniform, then we lower s_mul_u64 as it is.
9188 return Op;
9189}
9190
9191SDValue SITargetLowering::lowerXMULO(SDValue Op, SelectionDAG &DAG) const {
9192 EVT VT = Op.getValueType();
9193 SDLoc SL(Op);
9194 SDValue LHS = Op.getOperand(0);
9195 SDValue RHS = Op.getOperand(1);
9196 bool isSigned = Op.getOpcode() == ISD::SMULO;
9197
9198 if (ConstantSDNode *RHSC = isConstOrConstSplat(RHS)) {
9199 const APInt &C = RHSC->getAPIntValue();
9200 // mulo(X, 1 << S) -> { X << S, (X << S) >> S != X }
9201 if (C.isPowerOf2()) {
9202 // smulo(x, signed_min) is same as umulo(x, signed_min).
9203 bool UseArithShift = isSigned && !C.isMinSignedValue();
9204 SDValue ShiftAmt = DAG.getConstant(C.logBase2(), SL, MVT::i32);
9205 SDValue Result = DAG.getNode(ISD::SHL, SL, VT, LHS, ShiftAmt);
9206 SDValue Overflow =
9207 DAG.getSetCC(SL, MVT::i1,
9208 DAG.getNode(UseArithShift ? ISD::SRA : ISD::SRL, SL, VT,
9209 Result, ShiftAmt),
9210 LHS, ISD::SETNE);
9211 return DAG.getMergeValues({Result, Overflow}, SL);
9212 }
9213 }
9214
9215 SDValue Result = DAG.getNode(ISD::MUL, SL, VT, LHS, RHS);
9216 SDValue Top =
9217 DAG.getNode(isSigned ? ISD::MULHS : ISD::MULHU, SL, VT, LHS, RHS);
9218
9219 SDValue Sign = isSigned
9220 ? DAG.getNode(ISD::SRA, SL, VT, Result,
9221 DAG.getConstant(VT.getScalarSizeInBits() - 1,
9222 SL, MVT::i32))
9223 : DAG.getConstant(0, SL, VT);
9224 SDValue Overflow = DAG.getSetCC(SL, MVT::i1, Top, Sign, ISD::SETNE);
9225
9226 return DAG.getMergeValues({Result, Overflow}, SL);
9227}
9228
9229SDValue SITargetLowering::lowerXMUL_LOHI(SDValue Op, SelectionDAG &DAG) const {
9230 if (Op->isDivergent()) {
9231 // Select to V_MAD_[IU]64_[IU]32.
9232 return Op;
9233 }
9234 if (Subtarget->hasSMulHi()) {
9235 // Expand to S_MUL_I32 + S_MUL_HI_[IU]32.
9236 return SDValue();
9237 }
9238 // The multiply is uniform but we would have to use V_MUL_HI_[IU]32 to
9239 // calculate the high part, so we might as well do the whole thing with
9240 // V_MAD_[IU]64_[IU]32.
9241 return Op;
9242}
9243
9244SDValue SITargetLowering::lowerTRAP(SDValue Op, SelectionDAG &DAG) const {
9245 if (!Subtarget->hasTrapHandler() ||
9246 Subtarget->getTrapHandlerAbi() != GCNSubtarget::TrapHandlerAbi::AMDHSA)
9247 return lowerTrapEndpgm(Op, DAG);
9248
9249 return Subtarget->supportsGetDoorbellID() ? lowerTrapHsa(Op, DAG)
9250 : lowerTrapHsaQueuePtr(Op, DAG);
9251}
9252
9253SDValue SITargetLowering::lowerTrapEndpgm(SDValue Op, SelectionDAG &DAG) const {
9254 SDLoc SL(Op);
9255 SDValue Chain = Op.getOperand(0);
9256 return DAG.getNode(AMDGPUISD::ENDPGM_TRAP, SL, MVT::Other, Chain);
9257}
9258
9259SDValue
9260SITargetLowering::loadImplicitKernelArgument(SelectionDAG &DAG, MVT VT,
9261 const SDLoc &DL, Align Alignment,
9262 ImplicitParameter Param) const {
9265 SDValue Ptr = lowerKernArgParameterPtr(DAG, DL, DAG.getEntryNode(), Offset);
9266 MachinePointerInfo PtrInfo =
9268 return DAG.getLoad(
9269 VT, DL, DAG.getEntryNode(), Ptr, PtrInfo.getWithOffset(Offset), Alignment,
9271}
9272
9273SDValue SITargetLowering::lowerTrapHsaQueuePtr(SDValue Op,
9274 SelectionDAG &DAG) const {
9275 SDLoc SL(Op);
9276 SDValue Chain = Op.getOperand(0);
9277
9278 SDValue QueuePtr;
9279 // For code object version 5, QueuePtr is passed through implicit kernarg.
9280 const Module *M = DAG.getMachineFunction().getFunction().getParent();
9282 QueuePtr =
9283 loadImplicitKernelArgument(DAG, MVT::i64, SL, Align(8), QUEUE_PTR);
9284 } else {
9286 SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
9287 Register UserSGPR = Info->getQueuePtrUserSGPR();
9288
9289 if (UserSGPR == AMDGPU::NoRegister) {
9290 // We probably are in a function incorrectly marked with
9291 // amdgpu-no-queue-ptr. This is undefined. We don't want to delete the
9292 // trap, so just use a null pointer.
9293 QueuePtr = DAG.getConstant(0, SL, MVT::i64);
9294 } else {
9295 QueuePtr = CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass, UserSGPR,
9296 MVT::i64);
9297 }
9298 }
9299
9300 SDValue SGPR01 = DAG.getRegister(AMDGPU::SGPR0_SGPR1, MVT::i64);
9301 SDValue ToReg = DAG.getCopyToReg(Chain, SL, SGPR01, QueuePtr, SDValue());
9302
9304 SDValue Ops[] = {ToReg, DAG.getTargetConstant(TrapID, SL, MVT::i16), SGPR01,
9305 ToReg.getValue(1)};
9306 return DAG.getNode(AMDGPUISD::TRAP, SL, MVT::Other, Ops);
9307}
9308
9309SDValue SITargetLowering::lowerTrapHsa(SDValue Op, SelectionDAG &DAG) const {
9310 SDLoc SL(Op);
9311 SDValue Chain = Op.getOperand(0);
9312
9313 // We need to simulate the 's_trap 2' instruction on targets that run in
9314 // PRIV=1 (where it is treated as a nop).
9315 if (Subtarget->hasPrivEnabledTrap2NopBug())
9316 return DAG.getNode(AMDGPUISD::SIMULATED_TRAP, SL, MVT::Other, Chain);
9317
9319 SDValue Ops[] = {Chain, DAG.getTargetConstant(TrapID, SL, MVT::i16)};
9320 return DAG.getNode(AMDGPUISD::TRAP, SL, MVT::Other, Ops);
9321}
9322
9323SDValue SITargetLowering::lowerDEBUGTRAP(SDValue Op, SelectionDAG &DAG) const {
9324 SDLoc SL(Op);
9325 SDValue Chain = Op.getOperand(0);
9327
9328 if (!Subtarget->hasTrapHandler() ||
9329 Subtarget->getTrapHandlerAbi() != GCNSubtarget::TrapHandlerAbi::AMDHSA) {
9330 LLVMContext &Ctx = MF.getFunction().getContext();
9331 Ctx.diagnose(DiagnosticInfoUnsupported(MF.getFunction(),
9332 "debugtrap handler not supported",
9333 Op.getDebugLoc(), DS_Warning));
9334 return Chain;
9335 }
9336
9337 uint64_t TrapID =
9339 SDValue Ops[] = {Chain, DAG.getTargetConstant(TrapID, SL, MVT::i16)};
9340 return DAG.getNode(AMDGPUISD::TRAP, SL, MVT::Other, Ops);
9341}
9342
9343/// When a divergent value (in VGPR) is passed to an inline asm with an SGPR
9344/// constraint ('s'), we need to insert v_readfirstlane to move the value from
9345/// VGPR to SGPR. This is done by modifying the CopyToReg nodes in the glue
9346/// chain that feed into the INLINEASM node.
9347SDValue SITargetLowering::LowerINLINEASM(SDValue Op, SelectionDAG &DAG) const {
9348 unsigned NumOps = Op.getNumOperands();
9349
9350 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
9351 SmallSet<Register, 8> SGPRInputRegs;
9352
9353 unsigned NumVals = 0;
9354 for (unsigned I = InlineAsm::Op_FirstOperand; I < NumOps - 1;
9355 I += 1 + NumVals) {
9356 const InlineAsm::Flag Flags(Op.getConstantOperandVal(I));
9357 NumVals = Flags.getNumOperandRegisters();
9358
9359 unsigned RCID;
9360 bool IsSGPRInput = Flags.getKind() == InlineAsm::Kind::RegUse &&
9361 NumVals > 0 && Flags.hasRegClassConstraint(RCID) &&
9362 TRI->isSGPRClass(TRI->getRegClass(RCID));
9363
9364 for (unsigned J = 0; J < NumVals; ++J) {
9365 SDValue Val = Op.getOperand(I + 1 + J);
9366 if (const RegisterSDNode *RegNode =
9368 Register Reg = RegNode->getReg();
9369 if (IsSGPRInput || (Reg.isPhysical() && TRI->isSGPRPhysReg(Reg)))
9370 SGPRInputRegs.insert(Reg);
9371 }
9372 }
9373 }
9374
9375 if (SGPRInputRegs.empty())
9376 return Op;
9377
9378 // Walk the glue chain and insert readfirstlane for divergent SGPR inputs.
9379 SDLoc DL(Op);
9380 SDNode *N = Op.getOperand(NumOps - 1).getNode();
9381
9382 while (N && N->getOpcode() == ISD::CopyToReg) {
9383 Register Reg = cast<RegisterSDNode>(N->getOperand(1))->getReg();
9384 SDValue SrcVal = N->getOperand(2);
9385
9386 // Insert readfirstlane if copying a divergent value to an SGPR input.
9387 if (SrcVal->isDivergent() && SGPRInputRegs.count(Reg)) {
9388 SDValue ReadFirstLaneID =
9389 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
9390 SDValue ReadFirstLane =
9392 ReadFirstLaneID, SrcVal);
9393
9394 SmallVector<SDValue, 4> Ops = {N->getOperand(0), N->getOperand(1),
9395 ReadFirstLane};
9396 if (N->getNumOperands() > 3)
9397 Ops.push_back(N->getOperand(3)); // Glue input
9398
9399 DAG.UpdateNodeOperands(N, Ops);
9400 }
9401
9402 // Follow glue chain to next CopyToReg.
9403 SDNode *Next = nullptr;
9404 for (unsigned I = 0, E = N->getNumOperands(); I != E; ++I) {
9405 if (N->getOperand(I).getValueType() == MVT::Glue) {
9406 Next = N->getOperand(I).getNode();
9407 break;
9408 }
9409 }
9410 N = Next;
9411 }
9412
9413 return Op;
9414}
9415
9416SDValue SITargetLowering::getSegmentAperture(unsigned AS, const SDLoc &DL,
9417 SelectionDAG &DAG) const {
9418 unsigned BaseAS = AS;
9419 unsigned SANum = AMDGPU::getSyntheticApertureNumber(AS);
9421 BaseAS = AMDGPUAS::LOCAL_ADDRESS;
9422
9423 SDValue Aperture = getBaseSegmentAperture(BaseAS, DL, DAG);
9424
9425 if (SANum != AMDGPU::SyntheticAperture::None) {
9426 SDValue Tag = DAG.getConstant(SANum, DL, MVT::i32);
9427 return DAG.getNode(ISD::OR, DL, MVT::i32, Aperture, Tag);
9428 }
9429
9430 return Aperture;
9431}
9432
9433SDValue SITargetLowering::getBaseSegmentAperture(unsigned AS, const SDLoc &DL,
9434 SelectionDAG &DAG) const {
9435 const bool IsLDS = (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::BARRIER);
9436
9437 if (Subtarget->hasApertureRegs()) {
9438 const unsigned ApertureRegNo =
9439 IsLDS ? AMDGPU::SRC_SHARED_BASE : AMDGPU::SRC_PRIVATE_BASE;
9440 assert((ApertureRegNo != AMDGPU::SRC_PRIVATE_BASE ||
9441 !Subtarget->hasGloballyAddressableScratch()) &&
9442 "Cannot use src_private_base with globally addressable scratch!");
9443 // Note: this feature (register) is broken. When used as a 32-bit operand,
9444 // it returns a wrong value (all zeroes?). The real value is in the upper 32
9445 // bits.
9446 //
9447 // To work around the issue, emit a 64 bit copy from this register
9448 // then extract the high bits. Note that this shouldn't even result in a
9449 // shift being emitted and simply become a pair of registers (e.g.):
9450 // s_mov_b64 s[6:7], src_shared_base
9451 // v_mov_b32_e32 v1, s7
9452 SDValue Copy =
9453 DAG.getCopyFromReg(DAG.getEntryNode(), DL, ApertureRegNo, MVT::v2i32);
9454 return DAG.getExtractVectorElt(DL, MVT::i32, Copy, 1);
9455 }
9456
9457 // For code object version 5, private_base and shared_base are passed through
9458 // implicit kernargs.
9459 const Module *M = DAG.getMachineFunction().getFunction().getParent();
9462 return loadImplicitKernelArgument(DAG, MVT::i32, DL, Align(4), Param);
9463 }
9464
9466 SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
9467 Register UserSGPR = Info->getQueuePtrUserSGPR();
9468 if (UserSGPR == AMDGPU::NoRegister) {
9469 // We probably are in a function incorrectly marked with
9470 // amdgpu-no-queue-ptr. This is undefined.
9471 return DAG.getPOISON(MVT::i32);
9472 }
9473
9474 SDValue QueuePtr =
9475 CreateLiveInRegister(DAG, &AMDGPU::SReg_64RegClass, UserSGPR, MVT::i64);
9476
9477 // Offset into amd_queue_t for group_segment_aperture_base_hi /
9478 // private_segment_aperture_base_hi.
9479 uint32_t StructOffset = IsLDS ? 0x40 : 0x44;
9480
9481 SDValue Ptr =
9482 DAG.getObjectPtrOffset(DL, QueuePtr, TypeSize::getFixed(StructOffset));
9483
9484 // TODO: Use custom target PseudoSourceValue.
9485 // TODO: We should use the value from the IR intrinsic call, but it might not
9486 // be available and how do we get it?
9487 MachinePointerInfo PtrInfo(AMDGPUAS::CONSTANT_ADDRESS);
9488 return DAG.getLoad(MVT::i32, DL, QueuePtr.getValue(1), Ptr, PtrInfo,
9489 commonAlignment(Align(64), StructOffset),
9492}
9493
9494/// Return true if the value is a known valid address, such that a null check is
9495/// not necessary.
9497 const AMDGPUTargetMachine &TM, unsigned AddrSpace) {
9499 return true;
9500
9501 if (auto *ConstVal = dyn_cast<ConstantSDNode>(Val))
9502 return ConstVal->getSExtValue() != AMDGPU::getNullPointerValue(AddrSpace);
9503
9504 // TODO: Search through arithmetic, handle arguments and loads
9505 // marked nonnull.
9506 return false;
9507}
9508
9509SDValue SITargetLowering::lowerADDRSPACECAST(SDValue Op,
9510 SelectionDAG &DAG) const {
9511 SDLoc SL(Op);
9512
9513 const AMDGPUTargetMachine &TM =
9514 static_cast<const AMDGPUTargetMachine &>(getTargetMachine());
9515
9516 const auto *ASC = cast<AddrSpaceCastSDNode>(Op);
9517 unsigned SrcAS = ASC->getSrcAddressSpace();
9518 SDValue Src = ASC->getOperand(0);
9519 unsigned DestAS = ASC->getDestAddressSpace();
9520 bool IsNonNull = ASC->getFlags().hasNonNull();
9521
9522 SDValue FlatNullPtr = DAG.getConstant(0, SL, MVT::i64);
9523
9524 // flat -> local/private/barrier
9525 if (SrcAS == AMDGPUAS::FLAT_ADDRESS) {
9526 if (DestAS == AMDGPUAS::LOCAL_ADDRESS ||
9527 DestAS == AMDGPUAS::PRIVATE_ADDRESS || DestAS == AMDGPUAS::BARRIER) {
9528 SDValue Ptr = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Src);
9529
9530 if (DestAS == AMDGPUAS::PRIVATE_ADDRESS &&
9531 Subtarget->hasGloballyAddressableScratch()) {
9532 // flat -> private with globally addressable scratch: subtract
9533 // src_flat_scratch_base_lo.
9534 SDValue FlatScratchBaseLo(
9535 DAG.getMachineNode(
9536 AMDGPU::S_MOV_B32, SL, MVT::i32,
9537 DAG.getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_LO, MVT::i32)),
9538 0);
9539 Ptr = DAG.getNode(ISD::SUB, SL, MVT::i32, Ptr, FlatScratchBaseLo);
9540 }
9541
9542 if (IsNonNull || isKnownNonNull(Op, DAG, TM, SrcAS))
9543 return Ptr;
9544
9545 unsigned NullVal = AMDGPU::getNullPointerValue(DestAS);
9546 SDValue SegmentNullPtr = DAG.getConstant(NullVal, SL, MVT::i32);
9547 SDValue NonNull = DAG.getSetCC(SL, MVT::i1, Src, FlatNullPtr, ISD::SETNE);
9548
9549 return DAG.getNode(ISD::SELECT, SL, MVT::i32, NonNull, Ptr,
9550 SegmentNullPtr);
9551 }
9552 }
9553
9554 // local/private/barrier -> flat
9555 if (DestAS == AMDGPUAS::FLAT_ADDRESS) {
9556 if (SrcAS == AMDGPUAS::LOCAL_ADDRESS ||
9557 SrcAS == AMDGPUAS::PRIVATE_ADDRESS || SrcAS == AMDGPUAS::BARRIER) {
9558 SDValue CvtPtr;
9559 if (SrcAS == AMDGPUAS::PRIVATE_ADDRESS &&
9560 Subtarget->hasGloballyAddressableScratch()) {
9561 // For wave32: Addr = (TID[4:0] << 52) + FLAT_SCRATCH_BASE + privateAddr
9562 // For wave64: Addr = (TID[5:0] << 51) + FLAT_SCRATCH_BASE + privateAddr
9563 SDValue AllOnes = DAG.getSignedTargetConstant(-1, SL, MVT::i32);
9564 SDValue ThreadID = DAG.getConstant(0, SL, MVT::i32);
9565 ThreadID = DAG.getNode(
9566 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
9567 DAG.getTargetConstant(Intrinsic::amdgcn_mbcnt_lo, SL, MVT::i32),
9568 AllOnes, ThreadID);
9569 if (Subtarget->isWave64())
9570 ThreadID = DAG.getNode(
9571 ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32,
9572 DAG.getTargetConstant(Intrinsic::amdgcn_mbcnt_hi, SL, MVT::i32),
9573 AllOnes, ThreadID);
9574 SDValue ShAmt = DAG.getShiftAmountConstant(
9575 57 - 32 - Subtarget->getWavefrontSizeLog2(), MVT::i32, SL);
9576 SDValue SrcHi = DAG.getNode(ISD::SHL, SL, MVT::i32, ThreadID, ShAmt);
9577 CvtPtr = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, Src, SrcHi);
9578 CvtPtr = DAG.getNode(ISD::BITCAST, SL, MVT::i64, CvtPtr);
9579 // Accessing src_flat_scratch_base_lo as a 64-bit operand gives the full
9580 // 64-bit hi:lo value.
9581 SDValue FlatScratchBase = {
9582 DAG.getMachineNode(
9583 AMDGPU::S_MOV_B64, SL, MVT::i64,
9584 DAG.getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE, MVT::i64)),
9585 0};
9586 CvtPtr = DAG.getNode(ISD::ADD, SL, MVT::i64, CvtPtr, FlatScratchBase);
9587 } else {
9588 SDValue Aperture = getSegmentAperture(SrcAS, SL, DAG);
9589
9590 CvtPtr = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, Src, Aperture);
9591 CvtPtr = DAG.getNode(ISD::BITCAST, SL, MVT::i64, CvtPtr);
9592 }
9593
9594 if (IsNonNull || isKnownNonNull(Op, DAG, TM, SrcAS))
9595 return CvtPtr;
9596
9597 unsigned NullVal = AMDGPU::getNullPointerValue(SrcAS);
9598 SDValue SegmentNullPtr = DAG.getConstant(NullVal, SL, MVT::i32);
9599
9600 SDValue NonNull =
9601 DAG.getSetCC(SL, MVT::i1, Src, SegmentNullPtr, ISD::SETNE);
9602
9603 return DAG.getNode(ISD::SELECT, SL, MVT::i64, NonNull, CvtPtr,
9604 FlatNullPtr);
9605 }
9606 }
9607
9608 if (SrcAS == AMDGPUAS::CONSTANT_ADDRESS_32BIT &&
9609 Op.getValueType() == MVT::i64) {
9610 const SIMachineFunctionInfo *Info =
9611 DAG.getMachineFunction().getInfo<SIMachineFunctionInfo>();
9612 if (Info->get32BitAddressHighBits() == 0)
9613 return DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i64, Src);
9614
9615 SDValue Hi = DAG.getConstant(Info->get32BitAddressHighBits(), SL, MVT::i32);
9616 SDValue Vec = DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, Src, Hi);
9617 return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
9618 }
9619
9620 if (DestAS == AMDGPUAS::CONSTANT_ADDRESS_32BIT &&
9621 Src.getValueType() == MVT::i64)
9622 return DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Src);
9623
9624 // global <-> flat are no-ops and never emitted.
9625
9626 // Invalid casts are poison.
9627 return DAG.getPOISON(Op->getValueType(0));
9628}
9629
9630// This lowers an INSERT_SUBVECTOR by extracting the individual elements from
9631// the small vector and inserting them into the big vector. That is better than
9632// the default expansion of doing it via a stack slot. Even though the use of
9633// the stack slot would be optimized away afterwards, the stack slot itself
9634// remains.
9635SDValue SITargetLowering::lowerINSERT_SUBVECTOR(SDValue Op,
9636 SelectionDAG &DAG) const {
9637 SDValue Vec = Op.getOperand(0);
9638 SDValue Ins = Op.getOperand(1);
9639 SDValue Idx = Op.getOperand(2);
9640 EVT VecVT = Vec.getValueType();
9641 EVT InsVT = Ins.getValueType();
9642 EVT EltVT = VecVT.getVectorElementType();
9643 unsigned InsNumElts = InsVT.getVectorNumElements();
9644 unsigned IdxVal = Idx->getAsZExtVal();
9645 SDLoc SL(Op);
9646
9647 if (EltVT.getScalarSizeInBits() == 16 && IdxVal % 2 == 0) {
9648 // Insert 32-bit registers at a time.
9649 assert(InsNumElts % 2 == 0 && "expect legal vector types");
9650
9651 unsigned VecNumElts = VecVT.getVectorNumElements();
9652 EVT NewVecVT =
9653 EVT::getVectorVT(*DAG.getContext(), MVT::i32, VecNumElts / 2);
9654 EVT NewInsVT = InsNumElts == 2 ? MVT::i32
9656 MVT::i32, InsNumElts / 2);
9657
9658 Vec = DAG.getNode(ISD::BITCAST, SL, NewVecVT, Vec);
9659 Ins = DAG.getNode(ISD::BITCAST, SL, NewInsVT, Ins);
9660
9661 for (unsigned I = 0; I != InsNumElts / 2; ++I) {
9662 SDValue Elt;
9663 if (InsNumElts == 2) {
9664 Elt = Ins;
9665 } else {
9666 Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Ins,
9667 DAG.getConstant(I, SL, MVT::i32));
9668 }
9669 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, SL, NewVecVT, Vec, Elt,
9670 DAG.getConstant(IdxVal / 2 + I, SL, MVT::i32));
9671 }
9672
9673 return DAG.getNode(ISD::BITCAST, SL, VecVT, Vec);
9674 }
9675
9676 for (unsigned I = 0; I != InsNumElts; ++I) {
9677 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT, Ins,
9678 DAG.getConstant(I, SL, MVT::i32));
9679 Vec = DAG.getNode(ISD::INSERT_VECTOR_ELT, SL, VecVT, Vec, Elt,
9680 DAG.getConstant(IdxVal + I, SL, MVT::i32));
9681 }
9682 return Vec;
9683}
9684
9685SDValue SITargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
9686 SelectionDAG &DAG) const {
9687 SDValue Vec = Op.getOperand(0);
9688 SDValue InsVal = Op.getOperand(1);
9689 SDValue Idx = Op.getOperand(2);
9690 EVT VecVT = Vec.getValueType();
9691 EVT EltVT = VecVT.getVectorElementType();
9692 unsigned VecSize = VecVT.getSizeInBits();
9693 unsigned EltSize = EltVT.getSizeInBits();
9694 SDLoc SL(Op);
9695
9696 // Specially handle the case of v4i16 with static indexing.
9697 unsigned NumElts = VecVT.getVectorNumElements();
9698 auto *KIdx = dyn_cast<ConstantSDNode>(Idx);
9699 if (NumElts == 4 && EltSize == 16 && KIdx) {
9700 SDValue BCVec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Vec);
9701
9702 SDValue LoHalf = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, BCVec,
9703 DAG.getConstant(0, SL, MVT::i32));
9704 SDValue HiHalf = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, BCVec,
9705 DAG.getConstant(1, SL, MVT::i32));
9706
9707 SDValue LoVec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i16, LoHalf);
9708 SDValue HiVec = DAG.getNode(ISD::BITCAST, SL, MVT::v2i16, HiHalf);
9709
9710 unsigned Idx = KIdx->getZExtValue();
9711 bool InsertLo = Idx < 2;
9712 SDValue InsHalf = DAG.getNode(
9713 ISD::INSERT_VECTOR_ELT, SL, MVT::v2i16, InsertLo ? LoVec : HiVec,
9714 DAG.getNode(ISD::BITCAST, SL, MVT::i16, InsVal),
9715 DAG.getConstant(InsertLo ? Idx : (Idx - 2), SL, MVT::i32));
9716
9717 InsHalf = DAG.getNode(ISD::BITCAST, SL, MVT::i32, InsHalf);
9718
9719 SDValue Concat =
9720 InsertLo ? DAG.getBuildVector(MVT::v2i32, SL, {InsHalf, HiHalf})
9721 : DAG.getBuildVector(MVT::v2i32, SL, {LoHalf, InsHalf});
9722
9723 return DAG.getNode(ISD::BITCAST, SL, VecVT, Concat);
9724 }
9725
9726 // Static indexing does not lower to stack access, and hence there is no need
9727 // for special custom lowering to avoid stack access.
9728 if (isa<ConstantSDNode>(Idx))
9729 return SDValue();
9730
9731 // Avoid stack access for dynamic indexing by custom lowering to
9732 // v_bfi_b32 (v_bfm_b32 16, (shl idx, 16)), val, vec
9733
9734 assert(VecSize <= 64 && "Expected target vector size to be <= 64 bits");
9735
9736 MVT IntVT = MVT::getIntegerVT(VecSize);
9737
9738 // Convert vector index to bit-index and get the required bit mask.
9739 assert(isPowerOf2_32(EltSize));
9740 const auto EltMask = maskTrailingOnes<uint64_t>(EltSize);
9741 SDValue ScaleFactor = DAG.getConstant(Log2_32(EltSize), SL, MVT::i32);
9742 SDValue ScaledIdx = DAG.getNode(ISD::SHL, SL, MVT::i32, Idx, ScaleFactor);
9743 SDValue BFM = DAG.getNode(ISD::SHL, SL, IntVT,
9744 DAG.getConstant(EltMask, SL, IntVT), ScaledIdx);
9745
9746 // 1. Create a congruent vector with the target value in each element.
9747 SDValue ExtVal = DAG.getNode(ISD::BITCAST, SL, IntVT,
9748 DAG.getSplatBuildVector(VecVT, SL, InsVal));
9749
9750 // 2. Mask off all other indices except the required index within (1).
9751 SDValue LHS = DAG.getNode(ISD::AND, SL, IntVT, BFM, ExtVal);
9752
9753 // 3. Mask off the required index within the target vector.
9754 SDValue BCVec = DAG.getNode(ISD::BITCAST, SL, IntVT, Vec);
9755 SDValue RHS =
9756 DAG.getNode(ISD::AND, SL, IntVT, DAG.getNOT(SL, BFM, IntVT), BCVec);
9757
9758 // 4. Get (2) and (3) ORed into the target vector.
9759 SDValue BFI =
9760 DAG.getNode(ISD::OR, SL, IntVT, LHS, RHS, SDNodeFlags::Disjoint);
9761
9762 return DAG.getNode(ISD::BITCAST, SL, VecVT, BFI);
9763}
9764
9765SDValue SITargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
9766 SelectionDAG &DAG) const {
9767 SDLoc SL(Op);
9768
9769 EVT ResultVT = Op.getValueType();
9770 SDValue Vec = Op.getOperand(0);
9771 SDValue Idx = Op.getOperand(1);
9772 EVT VecVT = Vec.getValueType();
9773 unsigned VecSize = VecVT.getSizeInBits();
9774 EVT EltVT = VecVT.getVectorElementType();
9775
9776 DAGCombinerInfo DCI(DAG, AfterLegalizeVectorOps, true, nullptr);
9777
9778 // Make sure we do any optimizations that will make it easier to fold
9779 // source modifiers before obscuring it with bit operations.
9780
9781 // XXX - Why doesn't this get called when vector_shuffle is expanded?
9782 if (SDValue Combined = performExtractVectorEltCombine(Op.getNode(), DCI))
9783 return Combined;
9784
9785 if (VecSize == 128 || VecSize == 256 || VecSize == 512) {
9786 SDValue Lo, Hi;
9787 auto [LoVT, HiVT] = DAG.GetSplitDestVTs(VecVT);
9788
9789 if (VecSize == 128) {
9790 SDValue V2 = DAG.getBitcast(MVT::v2i64, Vec);
9791 Lo = DAG.getBitcast(LoVT,
9792 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i64, V2,
9793 DAG.getConstant(0, SL, MVT::i32)));
9794 Hi = DAG.getBitcast(HiVT,
9795 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i64, V2,
9796 DAG.getConstant(1, SL, MVT::i32)));
9797 } else if (VecSize == 256) {
9798 SDValue V2 = DAG.getBitcast(MVT::v4i64, Vec);
9799 SDValue Parts[4];
9800 for (unsigned P = 0; P < 4; ++P) {
9801 Parts[P] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i64, V2,
9802 DAG.getConstant(P, SL, MVT::i32));
9803 }
9804
9805 Lo = DAG.getBitcast(LoVT, DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i64,
9806 Parts[0], Parts[1]));
9807 Hi = DAG.getBitcast(HiVT, DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i64,
9808 Parts[2], Parts[3]));
9809 } else {
9810 assert(VecSize == 512);
9811
9812 SDValue V2 = DAG.getBitcast(MVT::v8i64, Vec);
9813 SDValue Parts[8];
9814 for (unsigned P = 0; P < 8; ++P) {
9815 Parts[P] = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i64, V2,
9816 DAG.getConstant(P, SL, MVT::i32));
9817 }
9818
9819 Lo = DAG.getBitcast(LoVT,
9820 DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v4i64,
9821 Parts[0], Parts[1], Parts[2], Parts[3]));
9822 Hi = DAG.getBitcast(HiVT,
9823 DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v4i64,
9824 Parts[4], Parts[5], Parts[6], Parts[7]));
9825 }
9826
9827 EVT IdxVT = Idx.getValueType();
9828 unsigned NElem = VecVT.getVectorNumElements();
9829 assert(isPowerOf2_32(NElem));
9830 SDValue IdxMask = DAG.getConstant(NElem / 2 - 1, SL, IdxVT);
9831 SDValue NewIdx = DAG.getNode(ISD::AND, SL, IdxVT, Idx, IdxMask);
9832 SDValue Half = DAG.getSelectCC(SL, Idx, IdxMask, Hi, Lo, ISD::SETUGT);
9833 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT, Half, NewIdx);
9834 }
9835
9836 assert(VecSize <= 64);
9837
9838 MVT IntVT = MVT::getIntegerVT(VecSize);
9839
9840 // If Vec is just a SCALAR_TO_VECTOR, then use the scalar integer directly.
9841 SDValue VecBC = peekThroughBitcasts(Vec);
9842 if (VecBC.getOpcode() == ISD::SCALAR_TO_VECTOR) {
9843 SDValue Src = VecBC.getOperand(0);
9844 Src = DAG.getBitcast(Src.getValueType().changeTypeToInteger(), Src);
9845 Vec = DAG.getAnyExtOrTrunc(Src, SL, IntVT);
9846 }
9847
9848 unsigned EltSize = EltVT.getSizeInBits();
9849 assert(isPowerOf2_32(EltSize));
9850
9851 SDValue ScaleFactor = DAG.getConstant(Log2_32(EltSize), SL, MVT::i32);
9852
9853 // Convert vector index to bit-index (* EltSize)
9854 SDValue ScaledIdx = DAG.getNode(ISD::SHL, SL, MVT::i32, Idx, ScaleFactor);
9855
9856 SDValue BC = DAG.getNode(ISD::BITCAST, SL, IntVT, Vec);
9857 SDValue Elt = DAG.getNode(ISD::SRL, SL, IntVT, BC, ScaledIdx);
9858
9859 if (ResultVT == MVT::f16 || ResultVT == MVT::bf16) {
9860 SDValue Result = DAG.getNode(ISD::TRUNCATE, SL, MVT::i16, Elt);
9861 return DAG.getNode(ISD::BITCAST, SL, ResultVT, Result);
9862 }
9863
9864 return DAG.getAnyExtOrTrunc(Elt, SL, ResultVT);
9865}
9866
9867static bool elementPairIsContiguous(ArrayRef<int> Mask, int Elt) {
9868 assert(Elt % 2 == 0);
9869 return Mask[Elt + 1] == Mask[Elt] + 1 && (Mask[Elt] % 2 == 0);
9870}
9871
9872static bool elementPairIsOddToEven(ArrayRef<int> Mask, int Elt) {
9873 assert(Elt % 2 == 0);
9874 return Mask[Elt] >= 0 && Mask[Elt + 1] >= 0 && (Mask[Elt] & 1) &&
9875 !(Mask[Elt + 1] & 1);
9876}
9877
9878SDValue SITargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
9879 SelectionDAG &DAG) const {
9880 SDLoc SL(Op);
9881 EVT ResultVT = Op.getValueType();
9882 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(Op);
9883 MVT EltVT = ResultVT.getVectorElementType().getSimpleVT();
9884 const int NewSrcNumElts = 2;
9885 MVT PackVT = MVT::getVectorVT(EltVT, NewSrcNumElts);
9886 int SrcNumElts = Op.getOperand(0).getValueType().getVectorNumElements();
9887
9888 // Break up the shuffle into registers sized pieces.
9889 //
9890 // We're trying to form sub-shuffles that the register allocation pipeline
9891 // won't be able to figure out, like how to use v_pk_mov_b32 to do a register
9892 // blend or 16-bit op_sel. It should be able to figure out how to reassemble a
9893 // pair of copies into a consecutive register copy, so use the ordinary
9894 // extract_vector_elt lowering unless we can use the shuffle.
9895 //
9896 // TODO: This is a bit of hack, and we should probably always use
9897 // extract_subvector for the largest possible subvector we can (or at least
9898 // use it for PackVT aligned pieces). However we have worse support for
9899 // combines on them don't directly treat extract_subvector / insert_subvector
9900 // as legal. The DAG scheduler also ends up doing a worse job with the
9901 // extract_subvectors.
9902 const bool ShouldUseConsecutiveExtract = EltVT.getSizeInBits() == 16;
9903
9904 // vector_shuffle <0,1,6,7> lhs, rhs
9905 // -> concat_vectors (extract_subvector lhs, 0), (extract_subvector rhs, 2)
9906 //
9907 // vector_shuffle <6,7,2,3> lhs, rhs
9908 // -> concat_vectors (extract_subvector rhs, 2), (extract_subvector lhs, 2)
9909 //
9910 // vector_shuffle <6,7,0,1> lhs, rhs
9911 // -> concat_vectors (extract_subvector rhs, 2), (extract_subvector lhs, 0)
9912
9913 // Avoid scalarizing when both halves are reading from consecutive elements.
9914
9915 // If we're treating 2 element shuffles as legal, also create odd-to-even
9916 // shuffles of neighboring pairs.
9917 //
9918 // vector_shuffle <3,2,7,6> lhs, rhs
9919 // -> concat_vectors vector_shuffle <1, 0> (extract_subvector lhs, 0)
9920 // vector_shuffle <1, 0> (extract_subvector rhs, 2)
9921
9923 for (int I = 0, N = ResultVT.getVectorNumElements(); I != N; I += 2) {
9924 if (ShouldUseConsecutiveExtract &&
9926 const int Idx = SVN->getMaskElt(I);
9927 int VecIdx = Idx < SrcNumElts ? 0 : 1;
9928 int EltIdx = Idx < SrcNumElts ? Idx : Idx - SrcNumElts;
9929 SDValue SubVec = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, PackVT,
9930 SVN->getOperand(VecIdx),
9931 DAG.getConstant(EltIdx, SL, MVT::i32));
9932 Pieces.push_back(SubVec);
9933 } else if (elementPairIsOddToEven(SVN->getMask(), I) &&
9935 int Idx0 = SVN->getMaskElt(I);
9936 int Idx1 = SVN->getMaskElt(I + 1);
9937
9938 SDValue SrcOp0 = SVN->getOperand(0);
9939 SDValue SrcOp1 = SrcOp0;
9940 if (Idx0 >= SrcNumElts) {
9941 SrcOp0 = SVN->getOperand(1);
9942 Idx0 -= SrcNumElts;
9943 }
9944
9945 if (Idx1 >= SrcNumElts) {
9946 SrcOp1 = SVN->getOperand(1);
9947 Idx1 -= SrcNumElts;
9948 }
9949
9950 int AlignedIdx0 = Idx0 & ~(NewSrcNumElts - 1);
9951 int AlignedIdx1 = Idx1 & ~(NewSrcNumElts - 1);
9952
9953 // Extract nearest even aligned piece.
9954 SDValue SubVec0 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, PackVT, SrcOp0,
9955 DAG.getConstant(AlignedIdx0, SL, MVT::i32));
9956 SDValue SubVec1 = DAG.getNode(ISD::EXTRACT_SUBVECTOR, SL, PackVT, SrcOp1,
9957 DAG.getConstant(AlignedIdx1, SL, MVT::i32));
9958
9959 int NewMaskIdx0 = Idx0 - AlignedIdx0;
9960 int NewMaskIdx1 = Idx1 - AlignedIdx1;
9961
9962 SDValue Result0 = SubVec0;
9963 SDValue Result1 = SubVec0;
9964
9965 if (SubVec0 != SubVec1) {
9966 NewMaskIdx1 += NewSrcNumElts;
9967 Result1 = SubVec1;
9968 } else {
9969 Result1 = DAG.getPOISON(PackVT);
9970 }
9971
9972 SDValue Shuf = DAG.getVectorShuffle(PackVT, SL, Result0, Result1,
9973 {NewMaskIdx0, NewMaskIdx1});
9974 Pieces.push_back(Shuf);
9975 } else {
9976 const int Idx0 = SVN->getMaskElt(I);
9977 const int Idx1 = SVN->getMaskElt(I + 1);
9978 int VecIdx0 = Idx0 < SrcNumElts ? 0 : 1;
9979 int VecIdx1 = Idx1 < SrcNumElts ? 0 : 1;
9980 int EltIdx0 = Idx0 < SrcNumElts ? Idx0 : Idx0 - SrcNumElts;
9981 int EltIdx1 = Idx1 < SrcNumElts ? Idx1 : Idx1 - SrcNumElts;
9982
9983 SDValue Vec0 = SVN->getOperand(VecIdx0);
9984 SDValue Elt0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT, Vec0,
9985 DAG.getSignedConstant(EltIdx0, SL, MVT::i32));
9986
9987 SDValue Vec1 = SVN->getOperand(VecIdx1);
9988 SDValue Elt1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT, Vec1,
9989 DAG.getSignedConstant(EltIdx1, SL, MVT::i32));
9990 Pieces.push_back(DAG.getBuildVector(PackVT, SL, {Elt0, Elt1}));
9991 }
9992 }
9993
9994 return DAG.getNode(ISD::CONCAT_VECTORS, SL, ResultVT, Pieces);
9995}
9996
9997SDValue SITargetLowering::lowerSCALAR_TO_VECTOR(SDValue Op,
9998 SelectionDAG &DAG) const {
9999 SDValue SVal = Op.getOperand(0);
10000 EVT ResultVT = Op.getValueType();
10001 EVT SValVT = SVal.getValueType();
10002 SDValue UndefVal = DAG.getPOISON(SValVT);
10003 SDLoc SL(Op);
10004
10006 VElts.push_back(SVal);
10007 for (int I = 1, E = ResultVT.getVectorNumElements(); I < E; ++I)
10008 VElts.push_back(UndefVal);
10009
10010 return DAG.getBuildVector(ResultVT, SL, VElts);
10011}
10012
10013SDValue SITargetLowering::lowerBUILD_VECTOR(SDValue Op,
10014 SelectionDAG &DAG) const {
10015 SDLoc SL(Op);
10016 EVT VT = Op.getValueType();
10017
10018 if (VT == MVT::v2f16 || VT == MVT::v2i16 || VT == MVT::v2bf16) {
10019 assert(!Subtarget->hasVOP3PInsts() && "this should be legal");
10020
10021 SDValue Lo = Op.getOperand(0);
10022 SDValue Hi = Op.getOperand(1);
10023
10024 // Avoid adding defined bits with the zero_extend.
10025 if (Hi.isUndef()) {
10026 Lo = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Lo);
10027 SDValue ExtLo = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Lo);
10028 return DAG.getNode(ISD::BITCAST, SL, VT, ExtLo);
10029 }
10030
10031 Hi = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Hi);
10032 Hi = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Hi);
10033
10034 SDValue ShlHi = DAG.getNode(ISD::SHL, SL, MVT::i32, Hi,
10035 DAG.getConstant(16, SL, MVT::i32));
10036 if (Lo.isUndef())
10037 return DAG.getNode(ISD::BITCAST, SL, VT, ShlHi);
10038
10039 Lo = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Lo);
10040 Lo = DAG.getNode(ISD::ZERO_EXTEND, SL, MVT::i32, Lo);
10041
10042 SDValue Or =
10043 DAG.getNode(ISD::OR, SL, MVT::i32, Lo, ShlHi, SDNodeFlags::Disjoint);
10044 return DAG.getNode(ISD::BITCAST, SL, VT, Or);
10045 }
10046
10047 // Split into 2-element chunks.
10048 const unsigned NumParts = VT.getVectorNumElements() / 2;
10049 EVT PartVT = MVT::getVectorVT(VT.getVectorElementType().getSimpleVT(), 2);
10050 MVT PartIntVT = MVT::getIntegerVT(PartVT.getSizeInBits());
10051
10053 for (unsigned P = 0; P < NumParts; ++P) {
10054 SDValue Vec = DAG.getBuildVector(
10055 PartVT, SL, {Op.getOperand(P * 2), Op.getOperand(P * 2 + 1)});
10056 Casts.push_back(DAG.getNode(ISD::BITCAST, SL, PartIntVT, Vec));
10057 }
10058
10059 SDValue Blend =
10060 DAG.getBuildVector(MVT::getVectorVT(PartIntVT, NumParts), SL, Casts);
10061 return DAG.getNode(ISD::BITCAST, SL, VT, Blend);
10062}
10063
10065 const GlobalAddressSDNode *GA) const {
10066 // Named barriers have fixed, non-relocated LDS addresses, so a constant
10067 // offset into an array of them can be folded into the address.
10069 const auto *GV = dyn_cast<GlobalVariable>(GA->getGlobal());
10070 return GV && AMDGPU::isNamedBarrier(*GV);
10071 }
10072
10073 // OSes that use ELF REL relocations (instead of RELA) can only store a
10074 // 32-bit addend in the instruction, so it is not safe to allow offset folding
10075 // which can create arbitrary 64-bit addends. (This is only a problem for
10076 // R_AMDGPU_*32_HI relocations since other relocation types are unaffected by
10077 // the high 32 bits of the addend.)
10078 //
10079 // This should be kept in sync with how HasRelocationAddend is initialized in
10080 // the constructor of ELFAMDGPUAsmBackend.
10081 if (!Subtarget->isAmdHsaOS())
10082 return false;
10083
10084 // We can fold offsets for anything that doesn't require a GOT relocation.
10085 return (GA->getAddressSpace() == AMDGPUAS::GLOBAL_ADDRESS ||
10089}
10090
10091static SDValue
10093 const SDLoc &DL, int64_t Offset, EVT PtrVT,
10094 unsigned GAFlags = SIInstrInfo::MO_NONE) {
10095 assert(isInt<32>(Offset + 4) && "32-bit offset is expected!");
10096 // In order to support pc-relative addressing, the PC_ADD_REL_OFFSET SDNode is
10097 // lowered to the following code sequence:
10098 //
10099 // For constant address space:
10100 // s_getpc_b64 s[0:1]
10101 // s_add_u32 s0, s0, $symbol
10102 // s_addc_u32 s1, s1, 0
10103 //
10104 // s_getpc_b64 returns the address of the s_add_u32 instruction and then
10105 // a fixup or relocation is emitted to replace $symbol with a literal
10106 // constant, which is a pc-relative offset from the encoding of the $symbol
10107 // operand to the global variable.
10108 //
10109 // For global address space:
10110 // s_getpc_b64 s[0:1]
10111 // s_add_u32 s0, s0, $symbol@{gotpc}rel32@lo
10112 // s_addc_u32 s1, s1, $symbol@{gotpc}rel32@hi
10113 //
10114 // s_getpc_b64 returns the address of the s_add_u32 instruction and then
10115 // fixups or relocations are emitted to replace $symbol@*@lo and
10116 // $symbol@*@hi with lower 32 bits and higher 32 bits of a literal constant,
10117 // which is a 64-bit pc-relative offset from the encoding of the $symbol
10118 // operand to the global variable.
10119 if (((const GCNSubtarget &)DAG.getSubtarget()).has64BitLiterals()) {
10120 assert(GAFlags != SIInstrInfo::MO_NONE);
10121
10122 SDValue Ptr =
10123 DAG.getTargetGlobalAddress(GV, DL, MVT::i64, Offset, GAFlags + 2);
10124 return DAG.getNode(AMDGPUISD::PC_ADD_REL_OFFSET64, DL, PtrVT, Ptr);
10125 }
10126
10127 SDValue PtrLo = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, Offset, GAFlags);
10128 SDValue PtrHi;
10129 if (GAFlags == SIInstrInfo::MO_NONE)
10130 PtrHi = DAG.getTargetConstant(0, DL, MVT::i32);
10131 else
10132 PtrHi = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, Offset, GAFlags + 1);
10133 return DAG.getNode(AMDGPUISD::PC_ADD_REL_OFFSET, DL, PtrVT, PtrLo, PtrHi);
10134}
10135
10136SDValue SITargetLowering::LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI,
10137 SDValue Op,
10138 SelectionDAG &DAG) const {
10139 GlobalAddressSDNode *GSD = cast<GlobalAddressSDNode>(Op);
10140 SDLoc DL(GSD);
10141 EVT PtrVT = Op.getValueType();
10142
10143 const GlobalValue *GV = GSD->getGlobal();
10144 const unsigned AS = GSD->getAddressSpace();
10145 if (((AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::BARRIER) &&
10148 if (AS == AMDGPUAS::LOCAL_ADDRESS && GV->hasExternalLinkage()) {
10149 const GlobalVariable &GVar = *cast<GlobalVariable>(GV);
10150 // HIP uses an unsized array `extern __shared__ T s[]` or similar
10151 // zero-sized type in other languages to declare the dynamic shared
10152 // memory which size is not known at the compile time. They will be
10153 // allocated by the runtime and placed directly after the static
10154 // allocated ones. They all share the same offset.
10155 if (GVar.getGlobalSize(GVar.getDataLayout()) == 0) {
10156 assert(PtrVT == MVT::i32 && "32-bit pointer is expected.");
10157 // Adjust alignment for that dynamic shared memory array.
10159 MFI->setDynLDSAlign(F, GVar);
10160 MFI->setUsesDynamicLDS(true);
10161 return SDValue(
10162 DAG.getMachineNode(AMDGPU::GET_GROUPSTATICSIZE, DL, PtrVT), 0);
10163 }
10164 }
10166 }
10167
10168 if (AS == AMDGPUAS::BARRIER) {
10169 SDValue GA = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, GSD->getOffset(),
10171 return SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, GA), 0);
10172 }
10173
10174 if (AS == AMDGPUAS::LOCAL_ADDRESS) {
10175 SDValue GA = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, GSD->getOffset(),
10177 return DAG.getNode(AMDGPUISD::LDS, DL, MVT::i32, GA);
10178 }
10179
10180 if (Subtarget->isAmdPalOS() || Subtarget->isMesa3DOS()) {
10181 if (Subtarget->has64BitLiterals()) {
10182 SDValue Addr = DAG.getTargetGlobalAddress(
10183 GV, DL, MVT::i64, GSD->getOffset(), SIInstrInfo::MO_ABS64);
10184 return SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B64, DL, MVT::i64, Addr),
10185 0);
10186 }
10187
10188 SDValue AddrLo = DAG.getTargetGlobalAddress(
10189 GV, DL, MVT::i32, GSD->getOffset(), SIInstrInfo::MO_ABS32_LO);
10190 AddrLo = {DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, AddrLo), 0};
10191
10192 SDValue AddrHi = DAG.getTargetGlobalAddress(
10193 GV, DL, MVT::i32, GSD->getOffset(), SIInstrInfo::MO_ABS32_HI);
10194 AddrHi = {DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, AddrHi), 0};
10195
10196 return DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i64, AddrLo, AddrHi);
10197 }
10198
10199 if (shouldEmitFixup(GV))
10200 return buildPCRelGlobalAddress(DAG, GV, DL, GSD->getOffset(), PtrVT);
10201
10202 if (shouldEmitPCReloc(GV))
10203 return buildPCRelGlobalAddress(DAG, GV, DL, GSD->getOffset(), PtrVT,
10205
10206 SDValue GOTAddr = buildPCRelGlobalAddress(DAG, GV, DL, 0, PtrVT,
10208 PointerType *PtrTy =
10210 const DataLayout &DataLayout = DAG.getDataLayout();
10211 Align Alignment = DataLayout.getABITypeAlign(PtrTy);
10212 MachinePointerInfo PtrInfo =
10214
10215 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), GOTAddr, PtrInfo, Alignment,
10218}
10219
10220SDValue SITargetLowering::LowerExternalSymbol(SDValue Op,
10221 SelectionDAG &DAG) const {
10222 // TODO: Handle this. It should be mostly the same as LowerGlobalAddress.
10223 const Function &Fn = DAG.getMachineFunction().getFunction();
10224 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
10225 Fn, "unsupported external symbol", Op.getDebugLoc()));
10226 return DAG.getPOISON(Op.getValueType());
10227}
10228
10230 const SDLoc &DL, SDValue V) const {
10231 // We can't use S_MOV_B32 directly, because there is no way to specify m0 as
10232 // the destination register.
10233 //
10234 // We can't use CopyToReg, because MachineCSE won't combine COPY instructions,
10235 // so we will end up with redundant moves to m0.
10236 //
10237 // We use a pseudo to ensure we emit s_mov_b32 with m0 as the direct result.
10238
10239 // A Null SDValue creates a glue result.
10240 SDNode *M0 = DAG.getMachineNode(AMDGPU::SI_INIT_M0, DL, MVT::Other, MVT::Glue,
10241 V, Chain);
10242 return SDValue(M0, 0);
10243}
10244
10245SDValue SITargetLowering::lowerImplicitZextParam(SelectionDAG &DAG, SDValue Op,
10246 MVT VT,
10247 unsigned Offset) const {
10248 SDLoc SL(Op);
10249 SDValue Param = lowerKernargMemParameter(
10250 DAG, MVT::i32, MVT::i32, SL, DAG.getEntryNode(), Offset, Align(4), false);
10251 // The local size values will have the hi 16-bits as zero.
10252 return DAG.getNode(ISD::AssertZext, SL, MVT::i32, Param,
10253 DAG.getValueType(VT));
10254}
10255
10257 EVT VT) {
10260 "non-hsa intrinsic with hsa target", DL.getDebugLoc()));
10261 return DAG.getPOISON(VT);
10262}
10263
10265 EVT VT) {
10268 "intrinsic not supported on subtarget", DL.getDebugLoc()));
10269 return DAG.getPOISON(VT);
10270}
10271
10273 ArrayRef<SDValue> Elts) {
10274 assert(!Elts.empty());
10275 MVT Type;
10276 unsigned NumElts = Elts.size();
10277
10278 if (NumElts <= 12) {
10279 Type = MVT::getVectorVT(MVT::f32, NumElts);
10280 } else {
10281 assert(Elts.size() <= 16);
10282 Type = MVT::v16f32;
10283 NumElts = 16;
10284 }
10285
10286 SmallVector<SDValue, 16> VecElts(NumElts);
10287 for (unsigned i = 0; i < Elts.size(); ++i) {
10288 SDValue Elt = Elts[i];
10289 if (Elt.getValueType() != MVT::f32)
10290 Elt = DAG.getBitcast(MVT::f32, Elt);
10291 VecElts[i] = Elt;
10292 }
10293 for (unsigned i = Elts.size(); i < NumElts; ++i)
10294 VecElts[i] = DAG.getPOISON(MVT::f32);
10295
10296 if (NumElts == 1)
10297 return VecElts[0];
10298 return DAG.getBuildVector(Type, DL, VecElts);
10299}
10300
10301static SDValue padEltsToUndef(SelectionDAG &DAG, const SDLoc &DL, EVT CastVT,
10302 SDValue Src, int ExtraElts) {
10303 EVT SrcVT = Src.getValueType();
10304
10306
10307 if (SrcVT.isVector())
10308 DAG.ExtractVectorElements(Src, Elts);
10309 else
10310 Elts.push_back(Src);
10311
10312 SDValue Undef = DAG.getPOISON(SrcVT.getScalarType());
10313 while (ExtraElts--)
10314 Elts.push_back(Undef);
10315
10316 return DAG.getBuildVector(CastVT, DL, Elts);
10317}
10318
10319// Re-construct the required return value for a image load intrinsic.
10320// This is more complicated due to the optional use TexFailCtrl which means the
10321// required return type is an aggregate
10323 ArrayRef<EVT> ResultTypes, bool IsTexFail,
10324 bool Unpacked, bool IsD16, int DMaskPop,
10325 int NumVDataDwords, bool IsAtomicPacked16Bit,
10326 const SDLoc &DL) {
10327 // Determine the required return type. This is the same regardless of
10328 // IsTexFail flag
10329 EVT ReqRetVT = ResultTypes[0];
10330 int ReqRetNumElts = ReqRetVT.isVector() ? ReqRetVT.getVectorNumElements() : 1;
10331 int NumDataDwords = ((IsD16 && !Unpacked) || IsAtomicPacked16Bit)
10332 ? (ReqRetNumElts + 1) / 2
10333 : ReqRetNumElts;
10334
10335 int MaskPopDwords = (!IsD16 || Unpacked) ? DMaskPop : (DMaskPop + 1) / 2;
10336
10337 MVT DataDwordVT =
10338 NumDataDwords == 1 ? MVT::i32 : MVT::getVectorVT(MVT::i32, NumDataDwords);
10339
10340 MVT MaskPopVT =
10341 MaskPopDwords == 1 ? MVT::i32 : MVT::getVectorVT(MVT::i32, MaskPopDwords);
10342
10343 SDValue Data(Result, 0);
10344 SDValue TexFail;
10345
10346 if (DMaskPop > 0 && Data.getValueType() != MaskPopVT) {
10347 SDValue ZeroIdx = DAG.getConstant(0, DL, MVT::i32);
10348 if (MaskPopVT.isVector()) {
10349 Data = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, MaskPopVT,
10350 SDValue(Result, 0), ZeroIdx);
10351 } else {
10352 Data = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MaskPopVT,
10353 SDValue(Result, 0), ZeroIdx);
10354 }
10355 }
10356
10357 if (DataDwordVT.isVector() && !IsAtomicPacked16Bit)
10358 Data = padEltsToUndef(DAG, DL, DataDwordVT, Data,
10359 NumDataDwords - MaskPopDwords);
10360
10361 if (IsD16)
10362 Data = adjustLoadValueTypeImpl(Data, ReqRetVT, DL, DAG, Unpacked);
10363
10364 EVT LegalReqRetVT = ReqRetVT;
10365 if (!ReqRetVT.isVector()) {
10366 if (!Data.getValueType().isInteger())
10367 Data = DAG.getNode(ISD::BITCAST, DL,
10368 Data.getValueType().changeTypeToInteger(), Data);
10369 Data = DAG.getNode(ISD::TRUNCATE, DL, ReqRetVT.changeTypeToInteger(), Data);
10370 } else {
10371 // We need to widen the return vector to a legal type
10372 if ((ReqRetVT.getVectorNumElements() % 2) == 1 &&
10373 ReqRetVT.getVectorElementType().getSizeInBits() == 16) {
10374 LegalReqRetVT =
10376 ReqRetVT.getVectorNumElements() + 1);
10377 }
10378 }
10379 Data = DAG.getNode(ISD::BITCAST, DL, LegalReqRetVT, Data);
10380
10381 if (IsTexFail) {
10382 TexFail =
10383 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, SDValue(Result, 0),
10384 DAG.getConstant(MaskPopDwords, DL, MVT::i32));
10385
10386 return DAG.getMergeValues({Data, TexFail, SDValue(Result, 1)}, DL);
10387 }
10388
10389 if (Result->getNumValues() == 1)
10390 return Data;
10391
10392 return DAG.getMergeValues({Data, SDValue(Result, 1)}, DL);
10393}
10394
10395static bool parseTexFail(SDValue TexFailCtrl, SelectionDAG &DAG, SDValue *TFE,
10396 SDValue *LWE, bool &IsTexFail) {
10397 auto *TexFailCtrlConst = cast<ConstantSDNode>(TexFailCtrl.getNode());
10398
10399 uint64_t Value = TexFailCtrlConst->getZExtValue();
10400 if (Value) {
10401 IsTexFail = true;
10402 }
10403
10404 SDLoc DL(TexFailCtrlConst);
10405 *TFE = DAG.getTargetConstant((Value & 0x1) ? 1 : 0, DL, MVT::i32);
10406 Value &= ~(uint64_t)0x1;
10407 *LWE = DAG.getTargetConstant((Value & 0x2) ? 1 : 0, DL, MVT::i32);
10408 Value &= ~(uint64_t)0x2;
10409
10410 return Value == 0;
10411}
10412
10414 MVT PackVectorVT,
10415 SmallVectorImpl<SDValue> &PackedAddrs,
10416 unsigned DimIdx, unsigned EndIdx,
10417 unsigned NumGradients) {
10418 SDLoc DL(Op);
10419 for (unsigned I = DimIdx; I < EndIdx; I++) {
10420 SDValue Addr = Op.getOperand(I);
10421
10422 // Gradients are packed with undef for each coordinate.
10423 // In <hi 16 bit>,<lo 16 bit> notation, the registers look like this:
10424 // 1D: undef,dx/dh; undef,dx/dv
10425 // 2D: dy/dh,dx/dh; dy/dv,dx/dv
10426 // 3D: dy/dh,dx/dh; undef,dz/dh; dy/dv,dx/dv; undef,dz/dv
10427 if (((I + 1) >= EndIdx) ||
10428 ((NumGradients / 2) % 2 == 1 && (I == DimIdx + (NumGradients / 2) - 1 ||
10429 I == DimIdx + NumGradients - 1))) {
10430 if (Addr.getValueType() != MVT::i16)
10431 Addr = DAG.getBitcast(MVT::i16, Addr);
10432 Addr = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, Addr);
10433 } else {
10434 Addr = DAG.getBuildVector(PackVectorVT, DL, {Addr, Op.getOperand(I + 1)});
10435 I++;
10436 }
10437 Addr = DAG.getBitcast(MVT::f32, Addr);
10438 PackedAddrs.push_back(Addr);
10439 }
10440}
10441
10442/// Emit a DiagnosticInfoUnsupported for an unsupported image intrinsic and
10443/// return poison values of \p ResultTypes, preserving the chain if present.
10445 ArrayRef<EVT> ResultTypes,
10446 const SDLoc &DL, const Twine &Msg) {
10448 DAG.getMachineFunction().getFunction(), Msg, DL.getDebugLoc()));
10449 return DAG.getErrorMergeValues(ResultTypes, Op.getOperand(0), DL);
10450}
10451
10452SDValue SITargetLowering::lowerImage(SDValue Op,
10454 SelectionDAG &DAG, bool WithChain) const {
10455 SDLoc DL(Op);
10457 const GCNSubtarget *ST = &MF.getSubtarget<GCNSubtarget>();
10458 unsigned IntrOpcode = Intr->BaseOpcode;
10459 // For image atomic: use no-return opcode if result is unused.
10460 if (Intr->AtomicNoRetBaseOpcode != Intr->BaseOpcode &&
10461 !Op.getNode()->hasAnyUseOfValue(0))
10462 IntrOpcode = Intr->AtomicNoRetBaseOpcode;
10463 const AMDGPU::MIMGBaseOpcodeInfo *BaseOpcode =
10465 const AMDGPU::MIMGDimInfo *DimInfo = AMDGPU::getMIMGDimInfo(Intr->Dim);
10466 bool IsGFX10Plus = AMDGPU::isGFX10Plus(*Subtarget);
10467 bool IsGFX11Plus = AMDGPU::isGFX11Plus(*Subtarget);
10468 bool IsGFX12Plus = AMDGPU::isGFX12Plus(*Subtarget);
10469 bool IsGFX13 = AMDGPU::isGFX13(*Subtarget);
10470
10471 SmallVector<EVT, 3> ResultTypes(Op->values());
10472 SmallVector<EVT, 3> OrigResultTypes(Op->values());
10473 if (BaseOpcode->NoReturn && BaseOpcode->Atomic)
10474 ResultTypes.erase(&ResultTypes[0]);
10475
10476 bool IsD16 = false;
10477 bool IsG16 = false;
10478 bool IsA16 = false;
10479 SDValue VData;
10480 int NumVDataDwords = 0;
10481 bool AdjustRetType = false;
10482 bool IsAtomicPacked16Bit = false;
10483
10484 // Offset of intrinsic arguments
10485 const unsigned ArgOffset = WithChain ? 2 : 1;
10486
10487 unsigned DMask;
10488 unsigned DMaskLanes = 0;
10489
10490 if (BaseOpcode->Atomic) {
10491 VData = Op.getOperand(2);
10492
10493 IsAtomicPacked16Bit =
10494 (IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_F16 ||
10495 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_F16_NORTN ||
10496 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_BF16 ||
10497 IntrOpcode == AMDGPU::IMAGE_ATOMIC_PK_ADD_BF16_NORTN);
10498
10499 if (!IsAtomicPacked16Bit && VData.getValueSizeInBits() != 32 &&
10500 VData.getValueSizeInBits() != 64) {
10501 return diagnoseUnsupportedImage(DAG, Op, OrigResultTypes, DL,
10502 "unsupported image atomic data type");
10503 }
10504
10505 bool Is64Bit = VData.getValueSizeInBits() == 64;
10506 if (BaseOpcode->AtomicX2) {
10507 SDValue VData2 = Op.getOperand(3);
10508 VData = DAG.getBuildVector(Is64Bit ? MVT::v2i64 : MVT::v2i32, DL,
10509 {VData, VData2});
10510 if (Is64Bit)
10511 VData = DAG.getBitcast(MVT::v4i32, VData);
10512
10513 if (!BaseOpcode->NoReturn)
10514 ResultTypes[0] = Is64Bit ? MVT::v2i64 : MVT::v2i32;
10515
10516 DMask = Is64Bit ? 0xf : 0x3;
10517 NumVDataDwords = Is64Bit ? 4 : 2;
10518 } else {
10519 DMask = Is64Bit ? 0x3 : 0x1;
10520 NumVDataDwords = Is64Bit ? 2 : 1;
10521 }
10522 } else {
10523 DMask = Op->getConstantOperandVal(ArgOffset + Intr->DMaskIndex);
10524 DMaskLanes = BaseOpcode->Gather4 ? 4 : llvm::popcount(DMask);
10525
10526 if (BaseOpcode->Store) {
10527 VData = Op.getOperand(2);
10528
10529 MVT StoreVT = VData.getSimpleValueType();
10530 MVT StoreScalarVT = StoreVT.getScalarType();
10531 if (StoreScalarVT != MVT::f16 && StoreScalarVT.getSizeInBits() != 32 &&
10532 StoreScalarVT.getSizeInBits() != 64) {
10533 return diagnoseUnsupportedImage(DAG, Op, OrigResultTypes, DL,
10534 "unsupported image store data type");
10535 }
10536 if (StoreScalarVT == MVT::f16) {
10537 if (!Subtarget->hasD16Images() || !BaseOpcode->HasD16)
10538 return Op; // D16 is unsupported for this instruction
10539
10540 IsD16 = true;
10541 VData = handleD16VData(VData, DAG, true);
10542 }
10543
10544 NumVDataDwords = (VData.getValueType().getSizeInBits() + 31) / 32;
10545 } else if (!BaseOpcode->NoReturn) {
10546 // Work out the num dwords based on the dmask popcount and underlying type
10547 // and whether packing is supported.
10548 MVT LoadVT = ResultTypes[0].getSimpleVT();
10549 MVT LoadScalarVT = LoadVT.getScalarType();
10550 if (LoadScalarVT != MVT::f16 && LoadScalarVT.getSizeInBits() != 32 &&
10551 LoadScalarVT.getSizeInBits() != 64) {
10552 return diagnoseUnsupportedImage(DAG, Op, OrigResultTypes, DL,
10553 "unsupported image load data type");
10554 }
10555 if (LoadScalarVT == MVT::f16) {
10556 if (!Subtarget->hasD16Images() || !BaseOpcode->HasD16)
10557 return Op; // D16 is unsupported for this instruction
10558
10559 IsD16 = true;
10560 }
10561
10562 // Confirm that the return type is large enough for the dmask specified
10563 if ((LoadVT.isVector() && LoadVT.getVectorNumElements() < DMaskLanes) ||
10564 (!LoadVT.isVector() && DMaskLanes > 1))
10565 return Op;
10566
10567 // The sq block of gfx8 and gfx9 do not estimate register use correctly
10568 // for d16 image_gather4, image_gather4_l, and image_gather4_lz
10569 // instructions.
10570 if (IsD16 && !Subtarget->hasUnpackedD16VMem() &&
10571 !(BaseOpcode->Gather4 && Subtarget->hasImageGather4D16Bug()))
10572 NumVDataDwords = (DMaskLanes + 1) / 2;
10573 else
10574 NumVDataDwords = DMaskLanes;
10575
10576 AdjustRetType = true;
10577 }
10578 }
10579
10580 unsigned VAddrEnd = ArgOffset + Intr->VAddrEnd;
10582
10583 // Check for 16 bit addresses or derivatives and pack if true.
10584 MVT VAddrVT =
10585 Op.getOperand(ArgOffset + Intr->GradientStart).getSimpleValueType();
10586 MVT VAddrScalarVT = VAddrVT.getScalarType();
10587 MVT GradPackVectorVT = VAddrScalarVT == MVT::f16 ? MVT::v2f16 : MVT::v2i16;
10588 IsG16 = VAddrScalarVT == MVT::f16 || VAddrScalarVT == MVT::i16;
10589
10590 VAddrVT = Op.getOperand(ArgOffset + Intr->CoordStart).getSimpleValueType();
10591 VAddrScalarVT = VAddrVT.getScalarType();
10592 MVT AddrPackVectorVT = VAddrScalarVT == MVT::f16 ? MVT::v2f16 : MVT::v2i16;
10593 IsA16 = VAddrScalarVT == MVT::f16 || VAddrScalarVT == MVT::i16;
10594
10595 // Push back extra arguments.
10596 for (unsigned I = Intr->VAddrStart; I < Intr->GradientStart; I++) {
10597 if (IsA16 && (Op.getOperand(ArgOffset + I).getValueType() == MVT::f16)) {
10598 assert(I == Intr->BiasIndex && "Got unexpected 16-bit extra argument");
10599 // Special handling of bias when A16 is on. Bias is of type half but
10600 // occupies full 32-bit.
10601 SDValue Bias = DAG.getBuildVector(
10602 MVT::v2f16, DL,
10603 {Op.getOperand(ArgOffset + I), DAG.getPOISON(MVT::f16)});
10604 VAddrs.push_back(Bias);
10605 } else {
10606 assert((!IsA16 || Intr->NumBiasArgs == 0 || I != Intr->BiasIndex) &&
10607 "Bias needs to be converted to 16 bit in A16 mode");
10608 VAddrs.push_back(Op.getOperand(ArgOffset + I));
10609 }
10610 }
10611
10612 if (BaseOpcode->Gradients && !ST->hasG16() && (IsA16 != IsG16)) {
10613 // 16 bit gradients are supported, but are tied to the A16 control
10614 // so both gradients and addresses must be 16 bit
10615 LLVM_DEBUG(
10616 dbgs() << "Failed to lower image intrinsic: 16 bit addresses "
10617 "require 16 bit args for both gradients and addresses");
10618 return Op;
10619 }
10620
10621 if (IsA16) {
10622 if (!ST->hasA16()) {
10623 LLVM_DEBUG(dbgs() << "Failed to lower image intrinsic: Target does not "
10624 "support 16 bit addresses\n");
10625 return Op;
10626 }
10627 }
10628
10629 // We've dealt with incorrect input so we know that if IsA16, IsG16
10630 // are set then we have to compress/pack operands (either address,
10631 // gradient or both)
10632 // In the case where a16 and gradients are tied (no G16 support) then we
10633 // have already verified that both IsA16 and IsG16 are true
10634 if (BaseOpcode->Gradients && IsG16 && ST->hasG16()) {
10635 // Activate g16
10636 const AMDGPU::MIMGG16MappingInfo *G16MappingInfo =
10638 IntrOpcode = G16MappingInfo->G16; // set new opcode to variant with _g16
10639 }
10640
10641 // Add gradients (packed or unpacked)
10642 if (IsG16) {
10643 // Pack the gradients
10644 // const int PackEndIdx = IsA16 ? VAddrEnd : (ArgOffset + Intr->CoordStart);
10645 packImage16bitOpsToDwords(DAG, Op, GradPackVectorVT, VAddrs,
10646 ArgOffset + Intr->GradientStart,
10647 ArgOffset + Intr->CoordStart, Intr->NumGradients);
10648 } else {
10649 for (unsigned I = ArgOffset + Intr->GradientStart;
10650 I < ArgOffset + Intr->CoordStart; I++)
10651 VAddrs.push_back(Op.getOperand(I));
10652 }
10653
10654 // Add addresses (packed or unpacked)
10655 if (IsA16) {
10656 packImage16bitOpsToDwords(DAG, Op, AddrPackVectorVT, VAddrs,
10657 ArgOffset + Intr->CoordStart, VAddrEnd,
10658 0 /* No gradients */);
10659 } else {
10660 // Add uncompressed address
10661 for (unsigned I = ArgOffset + Intr->CoordStart; I < VAddrEnd; I++)
10662 VAddrs.push_back(Op.getOperand(I));
10663 }
10664
10665 // If the register allocator cannot place the address registers contiguously
10666 // without introducing moves, then using the non-sequential address encoding
10667 // is always preferable, since it saves VALU instructions and is usually a
10668 // wash in terms of code size or even better.
10669 //
10670 // However, we currently have no way of hinting to the register allocator that
10671 // MIMG addresses should be placed contiguously when it is possible to do so,
10672 // so force non-NSA for the common 2-address case as a heuristic.
10673 //
10674 // SIShrinkInstructions will convert NSA encodings to non-NSA after register
10675 // allocation when possible.
10676 //
10677 // Partial NSA is allowed on GFX11+ where the final register is a contiguous
10678 // set of the remaining addresses.
10679 const unsigned NSAMaxSize = ST->getNSAMaxSize(BaseOpcode->Sampler);
10680 const bool HasPartialNSAEncoding = ST->hasPartialNSAEncoding();
10681 const bool UseNSA = ST->hasNSAEncoding() &&
10682 VAddrs.size() >= ST->getNSAThreshold(MF) &&
10683 (VAddrs.size() <= NSAMaxSize || HasPartialNSAEncoding);
10684 const bool UsePartialNSA =
10685 UseNSA && HasPartialNSAEncoding && VAddrs.size() > NSAMaxSize;
10686
10687 SDValue VAddr;
10688 if (UsePartialNSA) {
10689 VAddr = getBuildDwordsVector(DAG, DL,
10690 ArrayRef(VAddrs).drop_front(NSAMaxSize - 1));
10691 } else if (!UseNSA) {
10692 VAddr = getBuildDwordsVector(DAG, DL, VAddrs);
10693 }
10694
10695 SDValue True = DAG.getTargetConstant(1, DL, MVT::i1);
10696 SDValue False = DAG.getTargetConstant(0, DL, MVT::i1);
10697 SDValue Unorm;
10698 if (!BaseOpcode->Sampler) {
10699 Unorm = True;
10700 } else {
10701 uint64_t UnormConst =
10702 Op.getConstantOperandVal(ArgOffset + Intr->UnormIndex);
10703
10704 Unorm = UnormConst ? True : False;
10705 }
10706
10707 SDValue TFE;
10708 SDValue LWE;
10709 SDValue TexFail = Op.getOperand(ArgOffset + Intr->TexFailCtrlIndex);
10710 bool IsTexFail = false;
10711 if (!parseTexFail(TexFail, DAG, &TFE, &LWE, IsTexFail))
10712 return Op;
10713
10714 if (IsTexFail) {
10715 if (!DMaskLanes) {
10716 // Expecting to get an error flag since TFC is on - and dmask is 0
10717 // Force dmask to be at least 1 otherwise the instruction will fail
10718 DMask = 0x1;
10719 DMaskLanes = 1;
10720 NumVDataDwords = 1;
10721 }
10722 NumVDataDwords += 1;
10723 AdjustRetType = true;
10724 }
10725
10726 // Has something earlier tagged that the return type needs adjusting
10727 // This happens if the instruction is a load or has set TexFailCtrl flags
10728 if (AdjustRetType) {
10729 // NumVDataDwords reflects the true number of dwords required in the return
10730 // type
10731 if (DMaskLanes == 0 && !BaseOpcode->Store) {
10732 // This is a no-op load. This can be eliminated
10733 SDValue Undef = DAG.getPOISON(Op.getValueType());
10734 if (isa<MemSDNode>(Op))
10735 return DAG.getMergeValues({Undef, Op.getOperand(0)}, DL);
10736 return Undef;
10737 }
10738
10739 EVT NewVT = NumVDataDwords > 1 ? EVT::getVectorVT(*DAG.getContext(),
10740 MVT::i32, NumVDataDwords)
10741 : MVT::i32;
10742
10743 ResultTypes[0] = NewVT;
10744 if (ResultTypes.size() == 3) {
10745 // Original result was aggregate type used for TexFailCtrl results
10746 // The actual instruction returns as a vector type which has now been
10747 // created. Remove the aggregate result.
10748 ResultTypes.erase(&ResultTypes[1]);
10749 }
10750 }
10751
10752 unsigned CPol = Op.getConstantOperandVal(ArgOffset + Intr->CachePolicyIndex);
10753 // Keep GLC only when the atomic's result is actually used.
10754 if (BaseOpcode->Atomic && !BaseOpcode->NoReturn)
10756 if (CPol & ~((IsGFX12Plus ? AMDGPU::CPol::ALL : AMDGPU::CPol::ALL_pregfx12) |
10758 return Op;
10759
10761 if (BaseOpcode->Store || BaseOpcode->Atomic)
10762 Ops.push_back(VData); // vdata
10763 if (UsePartialNSA) {
10764 append_range(Ops, ArrayRef(VAddrs).take_front(NSAMaxSize - 1));
10765 Ops.push_back(VAddr);
10766 } else if (UseNSA)
10767 append_range(Ops, VAddrs);
10768 else
10769 Ops.push_back(VAddr);
10770 SDValue Rsrc = Op.getOperand(ArgOffset + Intr->RsrcIndex);
10771 EVT RsrcVT = Rsrc.getValueType();
10772 if (RsrcVT != MVT::v4i32 && RsrcVT != MVT::v8i32)
10773 return Op;
10774 Ops.push_back(Rsrc);
10775 if (BaseOpcode->Sampler) {
10776 SDValue Samp = Op.getOperand(ArgOffset + Intr->SampIndex);
10777 if (Samp.getValueType() != MVT::v4i32)
10778 return Op;
10779 Ops.push_back(Samp);
10780 }
10781 Ops.push_back(DAG.getTargetConstant(DMask, DL, MVT::i32));
10782 if (IsGFX10Plus)
10783 Ops.push_back(DAG.getTargetConstant(DimInfo->Encoding, DL, MVT::i32));
10784 if (!IsGFX12Plus || BaseOpcode->Sampler || BaseOpcode->MSAA)
10785 Ops.push_back(Unorm);
10786 Ops.push_back(DAG.getTargetConstant(CPol, DL, MVT::i32));
10787 Ops.push_back(IsA16 && // r128, a16 for gfx9
10788 ST->hasFeature(AMDGPU::FeatureR128A16)
10789 ? True
10790 : False);
10791 if (IsGFX10Plus)
10792 Ops.push_back(IsA16 ? True : False);
10793
10794 if (!Subtarget->hasGFX90AInsts())
10795 Ops.push_back(TFE); // tfe
10796 else if (TFE->getAsZExtVal()) {
10797 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
10799 "TFE is not supported on this GPU", DL.getDebugLoc()));
10800 }
10801
10802 if (!IsGFX12Plus || BaseOpcode->Sampler || BaseOpcode->MSAA)
10803 Ops.push_back(LWE); // lwe
10804 if (!IsGFX10Plus)
10805 Ops.push_back(DimInfo->DA ? True : False);
10806 if (BaseOpcode->HasD16)
10807 Ops.push_back(IsD16 ? True : False);
10808 if (isa<MemSDNode>(Op))
10809 Ops.push_back(Op.getOperand(0)); // chain
10810
10811 int NumVAddrDwords =
10812 UseNSA ? VAddrs.size() : VAddr.getValueType().getSizeInBits() / 32;
10813 int Opcode = -1;
10814
10815 if (IsGFX13) {
10816 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx13,
10817 NumVDataDwords, NumVAddrDwords);
10818 } else if (IsGFX12Plus) {
10819 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx12,
10820 NumVDataDwords, NumVAddrDwords);
10821 } else if (IsGFX11Plus) {
10822 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
10823 UseNSA ? AMDGPU::MIMGEncGfx11NSA
10824 : AMDGPU::MIMGEncGfx11Default,
10825 NumVDataDwords, NumVAddrDwords);
10826 } else if (IsGFX10Plus) {
10827 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode,
10828 UseNSA ? AMDGPU::MIMGEncGfx10NSA
10829 : AMDGPU::MIMGEncGfx10Default,
10830 NumVDataDwords, NumVAddrDwords);
10831 } else {
10832 if (Subtarget->hasGFX90AInsts()) {
10833 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx90a,
10834 NumVDataDwords, NumVAddrDwords);
10835 if (Opcode == -1) {
10837 DAG, Op, OrigResultTypes, DL,
10838 "requested image instruction is not supported on this GPU");
10839 }
10840 }
10841 if (Opcode == -1 &&
10842 Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
10843 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx8,
10844 NumVDataDwords, NumVAddrDwords);
10845 if (Opcode == -1)
10846 Opcode = AMDGPU::getMIMGOpcode(IntrOpcode, AMDGPU::MIMGEncGfx6,
10847 NumVDataDwords, NumVAddrDwords);
10848 }
10849 if (Opcode == -1)
10850 return Op;
10851
10852 MachineSDNode *NewNode = DAG.getMachineNode(Opcode, DL, ResultTypes, Ops);
10853 if (auto *MemOp = dyn_cast<MemSDNode>(Op)) {
10854 MachineMemOperand *MemRef = MemOp->getMemOperand();
10855 DAG.setNodeMemRefs(NewNode, {MemRef});
10856 }
10857
10858 if (BaseOpcode->NoReturn) {
10859 if (BaseOpcode->Atomic)
10860 return DAG.getMergeValues(
10861 {DAG.getPOISON(OrigResultTypes[0]), SDValue(NewNode, 0)}, DL);
10862
10863 return SDValue(NewNode, 0);
10864 }
10865
10866 if (BaseOpcode->AtomicX2) {
10868 DAG.ExtractVectorElements(SDValue(NewNode, 0), Elt, 0, 1);
10869 return DAG.getMergeValues({Elt[0], SDValue(NewNode, 1)}, DL);
10870 }
10871
10872 return constructRetValue(DAG, NewNode, OrigResultTypes, IsTexFail,
10873 Subtarget->hasUnpackedD16VMem(), IsD16, DMaskLanes,
10874 NumVDataDwords, IsAtomicPacked16Bit, DL);
10875}
10876
10877SDValue SITargetLowering::lowerSBuffer(EVT VT, EVT MemVT, SDLoc DL,
10878 SDValue Chain, SDValue Rsrc,
10879 SDValue Offset, SDValue CachePolicy,
10880 SelectionDAG &DAG,
10881 MachineMemOperand *MMO) const {
10883 bool HasChainResult = MMO != nullptr;
10884
10885 // SBUFFER_LOAD only produces values that fill whole SGPRs, apart from the
10886 // subword loads below.
10887 bool IsSubwordLoad = (MemVT == MVT::i8 || MemVT == MVT::i16) &&
10888 Subtarget->hasScalarSubwordLoads();
10889 if ((!isTypeLegal(VT) || VT.getSizeInBits() % 32 != 0) && !IsSubwordLoad) {
10890 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
10891 MF.getFunction(), "unsupported s_buffer_load result type",
10892 DL.getDebugLoc()));
10893 EVT ResultTypes[] = {VT, MVT::Other};
10894 return DAG.getErrorMergeValues(
10895 ArrayRef(ResultTypes, HasChainResult ? 2 : 1), Chain, DL);
10896 }
10897
10898 if (!HasChainResult) {
10899 const DataLayout &DataLayout = DAG.getDataLayout();
10901 DataLayout.getABITypeAlign(MemVT.getTypeForEVT(*DAG.getContext()));
10902
10903 MMO = MF.getMachineMemOperand(MachinePointerInfo(),
10907 MemVT.getStoreSize(), Alignment);
10908 }
10909
10910 if (!Offset->isDivergent()) {
10911 SDValue Ops[] = {Chain, Rsrc, Offset, CachePolicy};
10912
10913 // Lower llvm.amdgcn.*s.buffer.load.{i,u}N intrinsics. First, generate
10914 // s_buffer_load_u* for signed and unsigned load instructions. Next, DAG
10915 // combiner tries to merge the s_buffer_load_uN with a sext instruction
10916 // (performSignExtendInRegCombine()) and it replaces s_buffer_load_uN with
10917 // s_buffer_load_iN.
10918 auto HandleScalarSubwordLoads = [&](unsigned Opcode) -> SDValue {
10919 SDValue BufferLoad = DAG.getMemIntrinsicNode(
10920 Opcode, DL, DAG.getVTList(MVT::i32, MVT::Other), Ops, MemVT, MMO);
10921 SDValue LoadVal = DAG.getAnyExtOrTrunc(
10922 DAG.getNode(ISD::TRUNCATE, DL, MemVT, BufferLoad), DL, VT);
10923 if (HasChainResult)
10924 return DAG.getMergeValues({LoadVal, BufferLoad.getValue(1)}, DL);
10925 return LoadVal;
10926 };
10927 if (MemVT == MVT::i8 && Subtarget->hasScalarSubwordLoads())
10928 return HandleScalarSubwordLoads(AMDGPUISD::SBUFFER_LOAD_UBYTE);
10929
10930 if (MemVT == MVT::i16 && Subtarget->hasScalarSubwordLoads())
10931 return HandleScalarSubwordLoads(AMDGPUISD::SBUFFER_LOAD_USHORT);
10932
10933 // Widen vec3 load to vec4. Only 32-bit elements have a vec4 pattern.
10934 if (VT.isVector() && VT.getVectorNumElements() == 3 &&
10935 VT.getVectorElementType().getSizeInBits() == 32 &&
10936 !Subtarget->hasScalarDwordx3Loads()) {
10937 EVT WidenedVT =
10939 auto WidenedOp = DAG.getMemIntrinsicNode(
10940 AMDGPUISD::SBUFFER_LOAD, DL, DAG.getVTList(WidenedVT, MVT::Other),
10941 Ops, WidenedVT,
10942 MF.getMachineMemOperand(MMO, 0, WidenedVT.getStoreSize()));
10943 auto Subvector = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, WidenedOp,
10944 DAG.getVectorIdxConstant(0, DL));
10945 if (HasChainResult)
10946 return DAG.getMergeValues({Subvector, WidenedOp.getValue(1)}, DL);
10947 return Subvector;
10948 }
10949
10950 return DAG.getMemIntrinsicNode(AMDGPUISD::SBUFFER_LOAD, DL,
10951 DAG.getVTList(VT, MVT::Other), Ops, MemVT,
10952 MMO);
10953 }
10954
10955 // We have a divergent offset. Emit a MUBUF buffer load instead. We can
10956 // assume that the buffer is unswizzled.
10957 SDValue Ops[] = {
10958 Chain, // Chain
10959 Rsrc, // rsrc
10960 DAG.getConstant(0, DL, MVT::i32), // vindex
10961 {}, // voffset
10962 {}, // soffset
10963 {}, // offset
10964 CachePolicy, // cachepolicy
10965 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
10966 };
10967 if ((MemVT == MVT::i8 || MemVT == MVT::i16) &&
10968 Subtarget->hasScalarSubwordLoads()) {
10969 setBufferOffsets(Offset, DAG, &Ops[3], Align(4));
10970 SDValue Load = handleByteShortBufferLoads(DAG, MemVT, DL, Ops, MMO);
10971 SDValue LoadVal = DAG.getAnyExtOrTrunc(Load.getOperand(0), DL, VT);
10972 if (HasChainResult)
10973 return DAG.getMergeValues({LoadVal, Load.getOperand(1)}, DL);
10974 return LoadVal;
10975 }
10976
10978 unsigned NumLoads = 1;
10979 MVT LoadVT = VT.getSimpleVT();
10980 unsigned NumElts = LoadVT.isVector() ? LoadVT.getVectorNumElements() : 1;
10981 assert((LoadVT.getScalarType() == MVT::i32 ||
10982 LoadVT.getScalarType() == MVT::f32));
10983
10984 if (NumElts == 8 || NumElts == 16) {
10985 NumLoads = NumElts / 4;
10986 LoadVT = MVT::getVectorVT(LoadVT.getScalarType(), 4);
10987 }
10988
10989 SDVTList VTList = DAG.getVTList({LoadVT, MVT::Other});
10990
10991 // Use the alignment to ensure that the required offsets will fit into the
10992 // immediate offsets.
10993 setBufferOffsets(Offset, DAG, &Ops[3],
10994 NumLoads > 1 ? Align(16 * NumLoads) : Align(4));
10995
10996 uint64_t InstOffset = Ops[5]->getAsZExtVal();
10997 unsigned LoadSize = LoadVT.getStoreSize();
10998 for (unsigned i = 0; i < NumLoads; ++i) {
10999 Ops[5] = DAG.getTargetConstant(InstOffset + 16 * i, DL, MVT::i32);
11000 MachineMemOperand *LoadMMO = MF.getMachineMemOperand(MMO, 16 * i, LoadSize);
11001 Loads.push_back(getMemIntrinsicNode(AMDGPUISD::BUFFER_LOAD, DL, VTList, Ops,
11002 LoadVT, LoadMMO, DAG));
11003 }
11004
11005 if (NumElts == 8 || NumElts == 16) {
11006 SDValue LoadVal = DAG.getNode(ISD::CONCAT_VECTORS, DL, VT, Loads);
11007 if (HasChainResult) {
11008 SmallVector<SDValue, 4> LoadChains;
11009 for (SDValue Load : Loads)
11010 LoadChains.push_back(Load.getValue(1));
11011 SDValue Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, LoadChains);
11012 return DAG.getMergeValues({LoadVal, Chain}, DL);
11013 }
11014 return LoadVal;
11015 }
11016
11017 return Loads[0];
11018}
11019
11020SDValue SITargetLowering::lowerWaveID(SelectionDAG &DAG, SDValue Op) const {
11021 // With architected SGPRs, waveIDinGroup is in TTMP8[29:25].
11022 if (!Subtarget->hasArchitectedSGPRs())
11023 return {};
11024 SDLoc SL(Op);
11025 MVT VT = MVT::i32;
11026 SDValue TTMP8 = DAG.getCopyFromReg(DAG.getEntryNode(), SL, AMDGPU::TTMP8, VT);
11027 return DAG.getNode(AMDGPUISD::BFE_U32, SL, VT, TTMP8,
11028 DAG.getConstant(25, SL, VT), DAG.getConstant(5, SL, VT));
11029}
11030
11031SDValue SITargetLowering::lowerConstHwRegRead(SelectionDAG &DAG, SDValue Op,
11032 AMDGPU::Hwreg::Id HwReg,
11033 unsigned LowBit,
11034 unsigned Width) const {
11035 SDLoc SL(Op);
11036 using namespace AMDGPU::Hwreg;
11037 return {DAG.getMachineNode(
11038 AMDGPU::S_GETREG_B32_const, SL, MVT::i32,
11039 DAG.getTargetConstant(HwregEncoding::encode(HwReg, LowBit, Width),
11040 SL, MVT::i32)),
11041 0};
11042}
11043
11044SDValue SITargetLowering::lowerWorkitemID(SelectionDAG &DAG, SDValue Op,
11045 unsigned Dim,
11046 const ArgDescriptor &Arg) const {
11047 SDLoc SL(Op);
11049 unsigned MaxID = Subtarget->getMaxWorkitemID(MF.getFunction(), Dim);
11050 if (MaxID == 0)
11051 return DAG.getConstant(0, SL, MVT::i32);
11052
11053 // It's undefined behavior if a function marked with the amdgpu-no-*
11054 // attributes uses the corresponding intrinsic.
11055 if (!Arg)
11056 return DAG.getPOISON(Op->getValueType(0));
11057
11058 SDValue Val = loadInputValue(DAG, &AMDGPU::VGPR_32RegClass, MVT::i32,
11059 SDLoc(DAG.getEntryNode()), Arg);
11060
11061 // Don't bother inserting AssertZext for packed IDs since we're emitting the
11062 // masking operations anyway.
11063 //
11064 // TODO: We could assert the top bit is 0 for the source copy.
11065 if (Arg.isMasked())
11066 return Val;
11067
11068 // Preserve the known bits after expansion to a copy.
11069 EVT SmallVT = EVT::getIntegerVT(*DAG.getContext(), llvm::bit_width(MaxID));
11070 return DAG.getNode(ISD::AssertZext, SL, MVT::i32, Val,
11071 DAG.getValueType(SmallVT));
11072}
11073
11074SDValue SITargetLowering::lowerFromFP8(SDValue Op, bool IsBF8,
11075 SelectionDAG &DAG) const {
11076 SDLoc SL(Op);
11077 SDValue Src = Op.getOperand(0);
11078 EVT DstVT = Op.getValueType();
11079 bool IsF16 = DstVT.getVectorElementType() == MVT::f16;
11080 assert((!IsF16 || Subtarget->hasFP8F16ConversionInsts()) &&
11081 "fp8/bf8 -> f16 conversion requires FP8F16ConversionInsts");
11082
11083 unsigned Opc;
11084 if (IsF16)
11085 Opc = IsBF8 ? AMDGPUISD::CVT_PK_F16_BF8 : AMDGPUISD::CVT_PK_F16_FP8;
11086 else
11087 Opc = IsBF8 ? AMDGPUISD::CVT_PK_F32_BF8 : AMDGPUISD::CVT_PK_F32_FP8;
11088
11089 // Pack the two i8 lanes into the integer type the packed HW node reads. The
11090 // f16 form takes i16 and the f32 form takes i32. v2i8 bitcasts to i16
11091 // directly and the f32 node reads the low half of an any-extended i32.
11092 EVT PackedVT =
11094 SDValue AsI16 = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Src);
11095 SDValue Packed = DAG.getAnyExtOrTrunc(AsI16, SL, PackedVT);
11096 return DAG.getNode(Opc, SL, DstVT, Packed);
11097}
11098
11099SDValue
11100SITargetLowering::LowerCONVERT_FROM_ARBITRARY_FP(SDValue Op,
11101 SelectionDAG &DAG) const {
11102 // Handle the OCP FP8 formats (E4M3FN, E5M2) and unsigned E5M3 on subtargets
11103 // with matching HW conversions. Other formats use the generic expansion.
11104 APFloatBase::Semantics FPSemantic =
11105 static_cast<APFloatBase::Semantics>(Op.getConstantOperandVal(1));
11106 const bool IsFP8 = FPSemantic == APFloatBase::S_Float8E4M3FN;
11107 const bool IsBF8 = FPSemantic == APFloatBase::S_Float8E5M2;
11108 const bool IsE5M3 = FPSemantic == APFloatBase::S_Float8E5M3FNU;
11109 const bool HasE5M3ConversionInsts =
11110 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
11111 const bool IsSupported = IsFP8 || IsBF8 || (IsE5M3 && HasE5M3ConversionInsts);
11112 if (!IsSupported)
11113 return SDValue();
11114
11115 EVT DstVT = Op.getValueType();
11116 // The custom action for a v2i8 source also reaches half conversions on
11117 // targets which only have FP8-to-f32 instructions.
11118 if (DstVT.getScalarType() == MVT::f16 &&
11119 !Subtarget->hasFP8F16ConversionInsts())
11120 return SDValue();
11121
11122 if (IsE5M3) {
11123 if (DstVT.getScalarType() != MVT::f32)
11124 return SDValue();
11125
11126 SDLoc SL(Op);
11127 SDValue Src = Op.getOperand(0);
11128 assert((!DstVT.isVector() || DstVT == MVT::v2f32) &&
11129 "only the v2f32 vector result is custom lowered");
11130
11131 if (DstVT.isVector())
11132 Src = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Src);
11133 Src = DAG.getAnyExtOrTrunc(Src, SL, MVT::i32);
11134
11135 auto ConvertByte = [&](unsigned ByteSel) {
11136 return DAG.getNode(AMDGPUISD::CVT_F32_FP8_E5M3, SL, MVT::f32, Src,
11137 DAG.getTargetConstant(ByteSel, SL, MVT::i32));
11138 };
11139
11140 if (!DstVT.isVector())
11141 return ConvertByte(0);
11142 return DAG.getBuildVector(DstVT, SL, {ConvertByte(0), ConvertByte(1)});
11143 }
11144
11145 if (!DstVT.isVector()) {
11146 SDValue Src = Op.getOperand(0);
11147 if (Src.getValueType() != MVT::i32) {
11148 SDLoc SL(Op);
11149 SDValue SrcI32 = DAG.getAnyExtOrTrunc(Src, SL, MVT::i32);
11150 return DAG.getNode(ISD::CONVERT_FROM_ARBITRARY_FP, SL, DstVT, SrcI32,
11151 Op.getOperand(1));
11152 }
11153 return Op;
11154 }
11155
11156 EVT EltVT = DstVT.getVectorElementType();
11157 if (EltVT == MVT::f16 || EltVT == MVT::f32)
11158 return lowerFromFP8(Op, IsBF8, DAG);
11159 return SDValue();
11160}
11161
11162SDValue SITargetLowering::lowerToFP8(SDValue Op, bool IsBF8, bool IsE5M3,
11163 SelectionDAG &DAG) const {
11164 SDLoc SL(Op);
11165 SDValue Src = Op.getOperand(0);
11166 EVT ResVT = Op.getValueType();
11167 bool IsF16 = Src.getValueType().getScalarType() == MVT::f16;
11168 assert((!IsF16 || Subtarget->hasF16FP8ConversionInsts()) &&
11169 "f16 -> fp8/bf8 conversion requires F16FP8ConversionInsts");
11170 assert((!ResVT.isVector() || ResVT == MVT::v2i8) &&
11171 "only the v2i8 vector result is custom lowered");
11172
11173 if (IsF16) {
11174 unsigned Opc =
11175 IsBF8 ? AMDGPUISD::CVT_PK_BF8_F16 : AMDGPUISD::CVT_PK_FP8_F16;
11176 SDValue Bytes = DAG.getNode(Opc, SL, MVT::i16, Src);
11177 return DAG.getNode(ISD::BITCAST, SL, ResVT, Bytes);
11178 }
11179
11180 unsigned Opc = IsBF8 ? AMDGPUISD::CVT_PK_BF8_F32
11181 : IsE5M3 ? AMDGPUISD::CVT_PK_FP8_F32_E5M3
11182 : AMDGPUISD::CVT_PK_FP8_F32;
11183 SDValue PoisonI32 = DAG.getPOISON(MVT::i32);
11184 SDValue WordSel = DAG.getTargetConstant(0, SL, MVT::i1);
11185
11186 if (!ResVT.isVector()) {
11187 // Convert one lane, the second is unused. Feed it the same source so the
11188 // instruction does not read an undefined register.
11189 SDValue Packed =
11190 DAG.getNode(Opc, SL, MVT::i32, Src, Src, PoisonI32, WordSel);
11191 return DAG.getAnyExtOrTrunc(Packed, SL, ResVT);
11192 }
11193
11194 SDValue A = DAG.getExtractVectorElt(SL, MVT::f32, Src, 0);
11195 SDValue B = DAG.getExtractVectorElt(SL, MVT::f32, Src, 1);
11196 SDValue Packed = DAG.getNode(Opc, SL, MVT::i32, A, B, PoisonI32, WordSel);
11197 SDValue Bytes = DAG.getNode(ISD::TRUNCATE, SL, MVT::i16, Packed);
11198 return DAG.getNode(ISD::BITCAST, SL, ResVT, Bytes);
11199}
11200
11201SDValue
11202SITargetLowering::LowerCONVERT_TO_ARBITRARY_FP(SDValue Op,
11203 SelectionDAG &DAG) const {
11204 // The OCP FP8 formats (E4M3FN, E5M2) and unsigned E5M3 map to HW conversions
11205 // on subtargets that support them. Everything else uses generic expansion.
11207 static_cast<APFloatBase::Semantics>(Op.getConstantOperandVal(1));
11208 const bool IsFP8 = Sem == APFloatBase::S_Float8E4M3FN;
11209 const bool IsBF8 = Sem == APFloatBase::S_Float8E5M2;
11210 const bool IsE5M3 = Sem == APFloatBase::S_Float8E5M3FNU;
11211 const bool HasE5M3ConversionInsts =
11212 Subtarget->hasFP8ConversionInsts() && Subtarget->hasFP8E5M3Insts();
11213 const bool IsSupported = IsFP8 || IsBF8 || (IsE5M3 && HasE5M3ConversionInsts);
11214 if (!IsSupported)
11215 return SDValue();
11216
11217 // The HW conversions only support nearest-even. The OCP conversions do not
11218 // saturate. The unsigned E5M3 conversion always clamps out-of-range inputs,
11219 // which also refines the non-saturating form where those inputs are poison.
11220 if (static_cast<RoundingMode>(Op.getConstantOperandVal(2)) !=
11222 return SDValue();
11223 if (!IsE5M3 && Op.getConstantOperandVal(3) != 0)
11224 return SDValue();
11225
11226 EVT SrcEltVT = Op.getOperand(0).getValueType().getScalarType();
11227 // The f32 form is built here rather than by a tablegen pattern because the
11228 // HW result is i32 while the node result is i16 after the i8 promotion.
11229 if (SrcEltVT == MVT::f32)
11230 return lowerToFP8(Op, IsBF8, IsE5M3, DAG);
11231 if (!IsE5M3 && SrcEltVT == MVT::f16 &&
11232 Subtarget->hasF16FP8ConversionInsts()) {
11233 // A scalar conversion is selected from the generic node by tablegen, only
11234 // the illegal v2i8 result type needs lowering here.
11235 if (!Op.getValueType().isVector())
11236 return Op;
11237 return lowerToFP8(Op, IsBF8, false, DAG);
11238 }
11239 return SDValue();
11240}
11241
11242SDValue SITargetLowering::LowerINTRINSIC_WO_CHAIN(SDValue Op,
11243 SelectionDAG &DAG) const {
11245 auto *MFI = MF.getInfo<SIMachineFunctionInfo>();
11246
11247 EVT VT = Op.getValueType();
11248 SDLoc DL(Op);
11249 unsigned IntrinsicID = Op.getConstantOperandVal(0);
11250
11251 // TODO: Should this propagate fast-math-flags?
11252
11253 switch (IntrinsicID) {
11254 case Intrinsic::amdgcn_wave_reduce_min:
11255 case Intrinsic::amdgcn_wave_reduce_umin:
11256 case Intrinsic::amdgcn_wave_reduce_fmin:
11257 case Intrinsic::amdgcn_wave_reduce_max:
11258 case Intrinsic::amdgcn_wave_reduce_umax:
11259 case Intrinsic::amdgcn_wave_reduce_fmax:
11260 case Intrinsic::amdgcn_wave_reduce_add:
11261 case Intrinsic::amdgcn_wave_reduce_fadd:
11262 case Intrinsic::amdgcn_wave_reduce_sub:
11263 case Intrinsic::amdgcn_wave_reduce_fsub:
11264 case Intrinsic::amdgcn_wave_reduce_and:
11265 case Intrinsic::amdgcn_wave_reduce_or:
11266 case Intrinsic::amdgcn_wave_reduce_xor: {
11267 EVT SrcVT = Op.getOperand(1).getValueType();
11268 if (SrcVT.getFixedSizeInBits() == 16) {
11269 bool IsFPOp = SrcVT.isFloatingPoint();
11270 bool NeedsSignExt = IntrinsicID == Intrinsic::amdgcn_wave_reduce_min ||
11271 IntrinsicID == Intrinsic::amdgcn_wave_reduce_max ||
11272 IntrinsicID == Intrinsic::amdgcn_wave_reduce_add ||
11273 IntrinsicID == Intrinsic::amdgcn_wave_reduce_sub;
11274 unsigned ExtOpc = IsFPOp ? ISD::FP_EXTEND
11275 : NeedsSignExt ? ISD::SIGN_EXTEND
11277 auto SrcType = IsFPOp ? MVT::f16 : MVT::i16;
11278 auto ExtType = IsFPOp ? MVT::f32 : MVT::i32;
11279 SDValue ExtendedSrc = DAG.getNode(ExtOpc, DL, ExtType, Op.getOperand(1));
11280 SDValue Strategy = Op.getOperand(2);
11281 SDValue Result = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, ExtType,
11282 Op.getOperand(0), ExtendedSrc, Strategy);
11283 if (IsFPOp)
11284 return DAG.getNode(ISD::FP_ROUND, DL, SrcType, Result,
11285 DAG.getTargetConstant(1, DL, MVT::i32));
11286 else
11287 return DAG.getNode(ISD::TRUNCATE, DL, SrcType, Result);
11288 }
11289 return SDValue();
11290 }
11291 case Intrinsic::amdgcn_implicit_buffer_ptr: {
11292 if (getSubtarget()->isAmdHsaOrMesa(MF.getFunction()))
11293 return emitNonHSAIntrinsicError(DAG, DL, VT);
11294 return getPreloadedValue(DAG, *MFI, VT,
11296 }
11297 case Intrinsic::amdgcn_dispatch_ptr:
11298 case Intrinsic::amdgcn_queue_ptr: {
11299 if (!Subtarget->isAmdHsaOrMesa(MF.getFunction())) {
11300 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
11301 MF.getFunction(), "unsupported hsa intrinsic without hsa target",
11302 DL.getDebugLoc()));
11303 return DAG.getPOISON(VT);
11304 }
11305
11306 auto RegID = IntrinsicID == Intrinsic::amdgcn_dispatch_ptr
11309 return getPreloadedValue(DAG, *MFI, VT, RegID);
11310 }
11311 case Intrinsic::amdgcn_implicitarg_ptr: {
11312 if (MFI->isEntryFunction())
11313 return getImplicitArgPtr(DAG, DL);
11314 return getPreloadedValue(DAG, *MFI, VT,
11316 }
11317 case Intrinsic::amdgcn_kernarg_segment_ptr: {
11318 if (!AMDGPU::isKernel(MF.getFunction())) {
11319 // This only makes sense to call in a kernel, so just lower to null.
11320 return DAG.getConstant(0, DL, VT);
11321 }
11322
11323 return getPreloadedValue(DAG, *MFI, VT,
11325 }
11326 case Intrinsic::amdgcn_dispatch_id: {
11327 return getPreloadedValue(DAG, *MFI, VT, AMDGPUFunctionArgInfo::DISPATCH_ID);
11328 }
11329 case Intrinsic::amdgcn_rcp:
11330 return DAG.getNode(AMDGPUISD::RCP, DL, VT, Op.getOperand(1));
11331 case Intrinsic::amdgcn_rsq:
11332 return DAG.getNode(AMDGPUISD::RSQ, DL, VT, Op.getOperand(1));
11333 case Intrinsic::amdgcn_rsq_legacy:
11334 if (Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
11335 return emitRemovedIntrinsicError(DAG, DL, VT);
11336 return SDValue();
11337 case Intrinsic::amdgcn_rcp_legacy:
11338 if (Subtarget->getGeneration() >= AMDGPUSubtarget::VOLCANIC_ISLANDS)
11339 return emitRemovedIntrinsicError(DAG, DL, VT);
11340 return DAG.getNode(AMDGPUISD::RCP_LEGACY, DL, VT, Op.getOperand(1));
11341 case Intrinsic::amdgcn_fma_legacy:
11342 case Intrinsic::amdgcn_sudot4:
11343 case Intrinsic::amdgcn_sudot8:
11344 case Intrinsic::amdgcn_tanh:
11345 return SDValue();
11346 case Intrinsic::amdgcn_rsq_clamp: {
11347 if (Subtarget->getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS)
11348 return DAG.getNode(AMDGPUISD::RSQ_CLAMP, DL, VT, Op.getOperand(1));
11349
11350 Type *Type = VT.getTypeForEVT(*DAG.getContext());
11351 APFloat Max = APFloat::getLargest(Type->getFltSemantics());
11352 APFloat Min = APFloat::getLargest(Type->getFltSemantics(), true);
11353
11354 SDValue Rsq = DAG.getNode(AMDGPUISD::RSQ, DL, VT, Op.getOperand(1));
11355 SDValue Tmp =
11356 DAG.getNode(ISD::FMINNUM, DL, VT, Rsq, DAG.getConstantFP(Max, DL, VT));
11357 return DAG.getNode(ISD::FMAXNUM, DL, VT, Tmp,
11358 DAG.getConstantFP(Min, DL, VT));
11359 }
11360 case Intrinsic::r600_read_ngroups_x:
11361 if (Subtarget->isAmdHsaOS())
11362 return emitNonHSAIntrinsicError(DAG, DL, VT);
11363
11364 return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
11366 false);
11367 case Intrinsic::r600_read_ngroups_y:
11368 if (Subtarget->isAmdHsaOS())
11369 return emitNonHSAIntrinsicError(DAG, DL, VT);
11370
11371 return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
11373 false);
11374 case Intrinsic::r600_read_ngroups_z:
11375 if (Subtarget->isAmdHsaOS())
11376 return emitNonHSAIntrinsicError(DAG, DL, VT);
11377
11378 return lowerKernargMemParameter(DAG, VT, VT, DL, DAG.getEntryNode(),
11380 false);
11381 case Intrinsic::r600_read_local_size_x:
11382 if (Subtarget->isAmdHsaOS())
11383 return emitNonHSAIntrinsicError(DAG, DL, VT);
11384
11385 return lowerImplicitZextParam(DAG, Op, MVT::i16,
11387 case Intrinsic::r600_read_local_size_y:
11388 if (Subtarget->isAmdHsaOS())
11389 return emitNonHSAIntrinsicError(DAG, DL, VT);
11390
11391 return lowerImplicitZextParam(DAG, Op, MVT::i16,
11393 case Intrinsic::r600_read_local_size_z:
11394 if (Subtarget->isAmdHsaOS())
11395 return emitNonHSAIntrinsicError(DAG, DL, VT);
11396
11397 return lowerImplicitZextParam(DAG, Op, MVT::i16,
11399 case Intrinsic::amdgcn_workgroup_id_x:
11400 return lowerWorkGroupId(DAG, *MFI, VT,
11404 case Intrinsic::amdgcn_workgroup_id_y:
11405 return lowerWorkGroupId(DAG, *MFI, VT,
11409 case Intrinsic::amdgcn_workgroup_id_z:
11410 return lowerWorkGroupId(DAG, *MFI, VT,
11414 case Intrinsic::amdgcn_cluster_id_x:
11415 return Subtarget->hasClusters()
11416 ? getPreloadedValue(DAG, *MFI, VT,
11418 : DAG.getPOISON(VT);
11419 case Intrinsic::amdgcn_cluster_id_y:
11420 return Subtarget->hasClusters()
11421 ? getPreloadedValue(DAG, *MFI, VT,
11423 : DAG.getPOISON(VT);
11424 case Intrinsic::amdgcn_cluster_id_z:
11425 return Subtarget->hasClusters()
11426 ? getPreloadedValue(DAG, *MFI, VT,
11428 : DAG.getPOISON(VT);
11429 case Intrinsic::amdgcn_cluster_workgroup_id_x:
11430 return Subtarget->hasClusters()
11431 ? getPreloadedValue(
11432 DAG, *MFI, VT,
11434 : DAG.getPOISON(VT);
11435 case Intrinsic::amdgcn_cluster_workgroup_id_y:
11436 return Subtarget->hasClusters()
11437 ? getPreloadedValue(
11438 DAG, *MFI, VT,
11440 : DAG.getPOISON(VT);
11441 case Intrinsic::amdgcn_cluster_workgroup_id_z:
11442 return Subtarget->hasClusters()
11443 ? getPreloadedValue(
11444 DAG, *MFI, VT,
11446 : DAG.getPOISON(VT);
11447 case Intrinsic::amdgcn_cluster_workgroup_flat_id:
11448 return Subtarget->hasClusters()
11449 ? lowerConstHwRegRead(DAG, Op, AMDGPU::Hwreg::ID_IB_STS2, 21, 4)
11450 : SDValue();
11451 case Intrinsic::amdgcn_cluster_workgroup_max_id_x:
11452 return Subtarget->hasClusters()
11453 ? getPreloadedValue(
11454 DAG, *MFI, VT,
11456 : DAG.getPOISON(VT);
11457 case Intrinsic::amdgcn_cluster_workgroup_max_id_y:
11458 return Subtarget->hasClusters()
11459 ? getPreloadedValue(
11460 DAG, *MFI, VT,
11462 : DAG.getPOISON(VT);
11463 case Intrinsic::amdgcn_cluster_workgroup_max_id_z:
11464 return Subtarget->hasClusters()
11465 ? getPreloadedValue(
11466 DAG, *MFI, VT,
11468 : DAG.getPOISON(VT);
11469 case Intrinsic::amdgcn_cluster_workgroup_max_flat_id:
11470 return Subtarget->hasClusters()
11471 ? getPreloadedValue(
11472 DAG, *MFI, VT,
11474 : DAG.getPOISON(VT);
11475 case Intrinsic::amdgcn_wave_id:
11476 return lowerWaveID(DAG, Op);
11477 case Intrinsic::amdgcn_lds_kernel_id: {
11478 if (MFI->isEntryFunction())
11479 return getLDSKernelId(DAG, DL);
11480 return getPreloadedValue(DAG, *MFI, VT,
11482 }
11483 case Intrinsic::amdgcn_workitem_id_x:
11484 return lowerWorkitemID(DAG, Op, 0, MFI->getArgInfo().WorkItemIDX);
11485 case Intrinsic::amdgcn_workitem_id_y:
11486 return lowerWorkitemID(DAG, Op, 1, MFI->getArgInfo().WorkItemIDY);
11487 case Intrinsic::amdgcn_workitem_id_z:
11488 return lowerWorkitemID(DAG, Op, 2, MFI->getArgInfo().WorkItemIDZ);
11489 case Intrinsic::amdgcn_wavefrontsize:
11490 return DAG.getConstant(MF.getSubtarget<GCNSubtarget>().getWavefrontSize(),
11491 SDLoc(Op), MVT::i32);
11492 case Intrinsic::amdgcn_s_buffer_load: {
11493 unsigned CPol = Op.getConstantOperandVal(3);
11494 // s_buffer_load, because of how it's optimized, can't be volatile
11495 // so reject ones with the volatile bit set.
11496 if (CPol & ~((Subtarget->getGeneration() >= AMDGPUSubtarget::GFX12)
11499 return Op;
11500 return lowerSBuffer(VT, VT, DL, DAG.getEntryNode(), Op.getOperand(1),
11501 Op.getOperand(2), Op.getOperand(3), DAG);
11502 }
11503 case Intrinsic::amdgcn_fdiv_fast:
11504 return lowerFDIV_FAST(Op, DAG);
11505 case Intrinsic::amdgcn_sin:
11506 return DAG.getNode(AMDGPUISD::SIN_HW, DL, VT, Op.getOperand(1));
11507
11508 case Intrinsic::amdgcn_cos:
11509 return DAG.getNode(AMDGPUISD::COS_HW, DL, VT, Op.getOperand(1));
11510
11511 case Intrinsic::amdgcn_mul_u24:
11512 return DAG.getNode(AMDGPUISD::MUL_U24, DL, VT, Op.getOperand(1),
11513 Op.getOperand(2));
11514 case Intrinsic::amdgcn_mul_i24:
11515 return DAG.getNode(AMDGPUISD::MUL_I24, DL, VT, Op.getOperand(1),
11516 Op.getOperand(2));
11517
11518 case Intrinsic::amdgcn_log_clamp: {
11519 if (Subtarget->getGeneration() < AMDGPUSubtarget::VOLCANIC_ISLANDS)
11520 return SDValue();
11521
11522 return emitRemovedIntrinsicError(DAG, DL, VT);
11523 }
11524 case Intrinsic::amdgcn_fract:
11525 return DAG.getNode(AMDGPUISD::FRACT, DL, VT, Op.getOperand(1));
11526
11527 case Intrinsic::amdgcn_class: {
11528 SDValue Src = Op.getOperand(1);
11529 EVT SrcVT = Src.getValueType();
11530 bool IsLegal = SrcVT == MVT::f32 || SrcVT == MVT::f64 ||
11531 (SrcVT == MVT::f16 && Subtarget->has16BitInsts());
11532 if (!IsLegal) {
11533 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
11535 "llvm.amdgcn.class only supports f16, f32, and f64",
11536 DL.getDebugLoc()));
11537 return DAG.getPOISON(VT);
11538 }
11539 return DAG.getNode(AMDGPUISD::FP_CLASS, DL, VT, Src, Op.getOperand(2));
11540 }
11541 case Intrinsic::amdgcn_div_fmas:
11542 return DAG.getNode(AMDGPUISD::DIV_FMAS, DL, VT, Op.getOperand(1),
11543 Op.getOperand(2), Op.getOperand(3), Op.getOperand(4));
11544
11545 case Intrinsic::amdgcn_div_fixup:
11546 return DAG.getNode(AMDGPUISD::DIV_FIXUP, DL, VT, Op.getOperand(1),
11547 Op.getOperand(2), Op.getOperand(3));
11548
11549 case Intrinsic::amdgcn_div_scale: {
11550 const ConstantSDNode *Param = cast<ConstantSDNode>(Op.getOperand(3));
11551
11552 // Translate to the operands expected by the machine instruction. The
11553 // first parameter must be the same as the first instruction.
11554 SDValue Numerator = Op.getOperand(1);
11555 SDValue Denominator = Op.getOperand(2);
11556
11557 // Note this order is opposite of the machine instruction's operations,
11558 // which is s0.f = Quotient, s1.f = Denominator, s2.f = Numerator. The
11559 // intrinsic has the numerator as the first operand to match a normal
11560 // division operation.
11561
11562 SDValue Src0 = Param->isAllOnes() ? Numerator : Denominator;
11563
11564 return DAG.getNode(AMDGPUISD::DIV_SCALE, DL, Op->getVTList(), Src0,
11565 Denominator, Numerator);
11566 }
11567 case Intrinsic::amdgcn_ballot:
11568 return lowerBALLOTIntrinsic(*this, Op.getNode(), DAG);
11569 case Intrinsic::amdgcn_fmed3:
11570 return DAG.getNode(AMDGPUISD::FMED3, DL, VT, Op.getOperand(1),
11571 Op.getOperand(2), Op.getOperand(3), Op->getFlags());
11572 case Intrinsic::amdgcn_fdot2:
11573 return DAG.getNode(AMDGPUISD::FDOT2, DL, VT, Op.getOperand(1),
11574 Op.getOperand(2), Op.getOperand(3), Op.getOperand(4));
11575 case Intrinsic::amdgcn_fmul_legacy:
11576 return DAG.getNode(AMDGPUISD::FMUL_LEGACY, DL, VT, Op.getOperand(1),
11577 Op.getOperand(2));
11578 case Intrinsic::amdgcn_sbfe:
11579 case Intrinsic::amdgcn_ubfe:
11580 return lowerBFEIntrinsic(Op, DAG, IntrinsicID);
11581 case Intrinsic::amdgcn_cvt_pkrtz:
11582 case Intrinsic::amdgcn_cvt_pknorm_i16:
11583 case Intrinsic::amdgcn_cvt_pknorm_u16:
11584 case Intrinsic::amdgcn_cvt_pk_i16:
11585 case Intrinsic::amdgcn_cvt_pk_u16: {
11586 // FIXME: Stop adding cast if v2f16/v2i16 are legal.
11587 EVT VT = Op.getValueType();
11588 unsigned Opcode;
11589
11590 if (IntrinsicID == Intrinsic::amdgcn_cvt_pkrtz)
11591 Opcode = AMDGPUISD::CVT_PKRTZ_F16_F32;
11592 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pknorm_i16)
11593 Opcode = AMDGPUISD::CVT_PKNORM_I16_F32;
11594 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pknorm_u16)
11595 Opcode = AMDGPUISD::CVT_PKNORM_U16_F32;
11596 else if (IntrinsicID == Intrinsic::amdgcn_cvt_pk_i16)
11597 Opcode = AMDGPUISD::CVT_PK_I16_I32;
11598 else
11599 Opcode = AMDGPUISD::CVT_PK_U16_U32;
11600
11601 if (isTypeLegal(VT))
11602 return DAG.getNode(Opcode, DL, VT, Op.getOperand(1), Op.getOperand(2));
11603
11604 SDValue Node =
11605 DAG.getNode(Opcode, DL, MVT::i32, Op.getOperand(1), Op.getOperand(2));
11606 return DAG.getNode(ISD::BITCAST, DL, VT, Node);
11607 }
11608 case Intrinsic::amdgcn_fmad_ftz:
11609 return DAG.getNode(AMDGPUISD::FMAD_FTZ, DL, VT, Op.getOperand(1),
11610 Op.getOperand(2), Op.getOperand(3));
11611
11612 case Intrinsic::amdgcn_if_break:
11613 return SDValue(DAG.getMachineNode(AMDGPU::SI_IF_BREAK, DL, VT,
11614 Op->getOperand(1), Op->getOperand(2)),
11615 0);
11616
11617 case Intrinsic::amdgcn_groupstaticsize: {
11619 if (OS == Triple::AMDHSA || OS == Triple::AMDPAL)
11620 return Op;
11621
11622 const Module *M = MF.getFunction().getParent();
11623 const GlobalValue *GV =
11624 Intrinsic::getDeclarationIfExists(M, Intrinsic::amdgcn_groupstaticsize);
11625 SDValue GA = DAG.getTargetGlobalAddress(GV, DL, MVT::i32, 0,
11627 return {DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, GA), 0};
11628 }
11629 case Intrinsic::amdgcn_is_shared:
11630 case Intrinsic::amdgcn_is_private: {
11631 SDLoc SL(Op);
11632 SDValue SrcVec =
11633 DAG.getNode(ISD::BITCAST, DL, MVT::v2i32, Op.getOperand(1));
11634 SDValue SrcHi = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, SrcVec,
11635 DAG.getConstant(1, SL, MVT::i32));
11636
11637 unsigned AS = (IntrinsicID == Intrinsic::amdgcn_is_shared)
11639 : AMDGPUAS::PRIVATE_ADDRESS;
11640 if (AS == AMDGPUAS::PRIVATE_ADDRESS &&
11641 Subtarget->hasGloballyAddressableScratch()) {
11642 SDValue FlatScratchBaseHi(
11643 DAG.getMachineNode(
11644 AMDGPU::S_MOV_B32, DL, MVT::i32,
11645 DAG.getRegister(AMDGPU::SRC_FLAT_SCRATCH_BASE_HI, MVT::i32)),
11646 0);
11647 // Test bits 63..58 against the aperture address.
11648 return DAG.getSetCC(
11649 SL, MVT::i1,
11650 DAG.getNode(ISD::XOR, SL, MVT::i32, SrcHi, FlatScratchBaseHi),
11651 DAG.getConstant(1u << 26, SL, MVT::i32), ISD::SETULT);
11652 }
11653
11654 SDValue Aperture = getSegmentAperture(AS, SL, DAG);
11655 return DAG.getSetCC(SL, MVT::i1, SrcHi, Aperture, ISD::SETEQ);
11656 }
11657 case Intrinsic::amdgcn_perm:
11658 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, Op.getOperand(1),
11659 Op.getOperand(2), Op.getOperand(3));
11660 case Intrinsic::amdgcn_reloc_constant: {
11661 Module *M = MF.getFunction().getParent();
11662 const MDNode *Metadata = cast<MDNodeSDNode>(Op.getOperand(1))->getMD();
11663 auto SymbolName = cast<MDString>(Metadata->getOperand(0))->getString();
11664 auto *RelocSymbol = cast<GlobalVariable>(
11665 M->getOrInsertGlobal(SymbolName, Type::getInt32Ty(M->getContext())));
11666 SDValue GA = DAG.getTargetGlobalAddress(RelocSymbol, DL, MVT::i32, 0,
11668 return {DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, GA), 0};
11669 }
11670 case Intrinsic::amdgcn_swmmac_f16_16x16x32_f16:
11671 case Intrinsic::amdgcn_swmmac_bf16_16x16x32_bf16:
11672 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf16:
11673 case Intrinsic::amdgcn_swmmac_f32_16x16x32_f16:
11674 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_fp8:
11675 case Intrinsic::amdgcn_swmmac_f32_16x16x32_fp8_bf8:
11676 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_fp8:
11677 case Intrinsic::amdgcn_swmmac_f32_16x16x32_bf8_bf8: {
11678 if (Op.getOperand(4).getValueType() == MVT::i32)
11679 return SDValue();
11680
11681 SDLoc SL(Op);
11682 auto IndexKeyi32 = DAG.getAnyExtOrTrunc(Op.getOperand(4), SL, MVT::i32);
11683 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, Op.getValueType(),
11684 Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11685 Op.getOperand(3), IndexKeyi32);
11686 }
11687 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_fp8:
11688 case Intrinsic::amdgcn_swmmac_f32_16x16x128_fp8_bf8:
11689 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_fp8:
11690 case Intrinsic::amdgcn_swmmac_f32_16x16x128_bf8_bf8:
11691 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_fp8:
11692 case Intrinsic::amdgcn_swmmac_f16_16x16x128_fp8_bf8:
11693 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_fp8:
11694 case Intrinsic::amdgcn_swmmac_f16_16x16x128_bf8_bf8: {
11695 if (Op.getOperand(4).getValueType() == MVT::i64)
11696 return SDValue();
11697
11698 SDLoc SL(Op);
11699 auto IndexKeyi64 =
11700 Op.getOperand(4).getValueType() == MVT::v2i32
11701 ? DAG.getBitcast(MVT::i64, Op.getOperand(4))
11702 : DAG.getAnyExtOrTrunc(Op.getOperand(4), SL, MVT::i64);
11703 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, Op.getValueType(),
11704 {Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11705 Op.getOperand(3), IndexKeyi64, Op.getOperand(5),
11706 Op.getOperand(6)});
11707 }
11708 case Intrinsic::amdgcn_swmmac_f16_16x16x64_f16:
11709 case Intrinsic::amdgcn_swmmac_bf16_16x16x64_bf16:
11710 case Intrinsic::amdgcn_swmmac_f32_16x16x64_bf16:
11711 case Intrinsic::amdgcn_swmmac_bf16f32_16x16x64_bf16:
11712 case Intrinsic::amdgcn_swmmac_f32_16x16x64_f16:
11713 case Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8: {
11714 EVT IndexKeyTy = IntrinsicID == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8
11715 ? MVT::i64
11716 : MVT::i32;
11717 if (Op.getOperand(6).getValueType() == IndexKeyTy)
11718 return SDValue();
11719
11720 SDLoc SL(Op);
11721 auto IndexKey =
11722 Op.getOperand(6).getValueType().isVector()
11723 ? DAG.getBitcast(IndexKeyTy, Op.getOperand(6))
11724 : DAG.getAnyExtOrTrunc(Op.getOperand(6), SL, IndexKeyTy);
11726 Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11727 Op.getOperand(3), Op.getOperand(4), Op.getOperand(5),
11728 IndexKey, Op.getOperand(7), Op.getOperand(8)};
11729 if (IntrinsicID == Intrinsic::amdgcn_swmmac_i32_16x16x128_iu8)
11730 Args.push_back(Op.getOperand(9));
11731 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, Op.getValueType(), Args);
11732 }
11733 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu4:
11734 case Intrinsic::amdgcn_swmmac_i32_16x16x32_iu8:
11735 case Intrinsic::amdgcn_swmmac_i32_16x16x64_iu4: {
11736 if (Op.getOperand(6).getValueType() == MVT::i32)
11737 return SDValue();
11738
11739 SDLoc SL(Op);
11740 auto IndexKeyi32 = DAG.getAnyExtOrTrunc(Op.getOperand(6), SL, MVT::i32);
11741 return DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, Op.getValueType(),
11742 {Op.getOperand(0), Op.getOperand(1), Op.getOperand(2),
11743 Op.getOperand(3), Op.getOperand(4), Op.getOperand(5),
11744 IndexKeyi32, Op.getOperand(7)});
11745 }
11746 case Intrinsic::amdgcn_wmma_scale_f32_16x16x128_f8f6f4:
11747 case Intrinsic::amdgcn_wmma_scale16_f32_16x16x128_f8f6f4: {
11748 unsigned AFmt = (unsigned)Op.getConstantOperandVal(1);
11749 unsigned BFmt = (unsigned)Op.getConstantOperandVal(3);
11750 unsigned AScaleFmt = (unsigned)Op.getConstantOperandVal(8);
11751 unsigned BScaleFmt = (unsigned)Op.getConstantOperandVal(11);
11752 if (!AMDGPU::isValidWMMAScaleFmtCombination(AFmt, AScaleFmt, BFmt,
11753 BScaleFmt)) {
11755 "invalid matrix and scale format combination in wmma call");
11756 Op->print(errs());
11757 errs() << '\n';
11758 }
11759 return SDValue();
11760 }
11761 case Intrinsic::amdgcn_readlane:
11762 case Intrinsic::amdgcn_readfirstlane:
11763 case Intrinsic::amdgcn_writelane:
11764 case Intrinsic::amdgcn_permlane16:
11765 case Intrinsic::amdgcn_permlanex16:
11766 case Intrinsic::amdgcn_permlane64:
11767 case Intrinsic::amdgcn_set_inactive:
11768 case Intrinsic::amdgcn_set_inactive_chain_arg:
11769 case Intrinsic::amdgcn_mov_dpp8:
11770 case Intrinsic::amdgcn_update_dpp:
11771 case Intrinsic::amdgcn_permlane_bcast:
11772 case Intrinsic::amdgcn_permlane_up:
11773 case Intrinsic::amdgcn_permlane_down:
11774 case Intrinsic::amdgcn_permlane_xor:
11775 return lowerLaneOp(*this, Op.getNode(), DAG);
11776 case Intrinsic::amdgcn_dead: {
11778 for (const EVT ValTy : Op.getNode()->values())
11779 Poisons.push_back(DAG.getPOISON(ValTy));
11780 return DAG.getMergeValues(Poisons, SDLoc(Op));
11781 }
11782 case Intrinsic::amdgcn_wave_shuffle:
11783 return lowerWaveShuffle(*this, Op.getNode(), DAG);
11784 default:
11785 if (const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
11787 return lowerImage(Op, ImageDimIntr, DAG, false);
11788
11789 return Op;
11790 }
11791}
11792
11793// On targets not supporting constant in soffset field, turn zero to
11794// SGPR_NULL to avoid generating an extra s_mov with zero.
11796 const GCNSubtarget *Subtarget) {
11797 if (Subtarget->hasRestrictedSOffset() && isNullConstant(SOffset))
11798 return DAG.getRegister(AMDGPU::SGPR_NULL, MVT::i32);
11799 return SOffset;
11800}
11801
11802SDValue SITargetLowering::lowerRawBufferAtomicIntrin(SDValue Op,
11803 SelectionDAG &DAG,
11804 unsigned NewOpcode) const {
11805 SDLoc DL(Op);
11806
11807 SDValue VData = Op.getOperand(2);
11808 if (VData.getValueSizeInBits() != 32 && VData.getValueSizeInBits() != 64) {
11809 SmallVector<EVT, 2> ResultTypes(Op->values());
11810 return diagnoseUnsupportedImage(DAG, Op, ResultTypes, DL,
11811 "unsupported buffer atomic data type");
11812 }
11813 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
11814 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(4), DAG);
11815 auto SOffset = selectSOffset(Op.getOperand(5), DAG, Subtarget);
11816 SDValue Ops[] = {
11817 Op.getOperand(0), // Chain
11818 VData, // vdata
11819 Rsrc, // rsrc
11820 DAG.getConstant(0, DL, MVT::i32), // vindex
11821 VOffset, // voffset
11822 SOffset, // soffset
11823 Offset, // offset
11824 Op.getOperand(6), // cachepolicy
11825 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
11826 };
11827
11828 auto *M = cast<MemSDNode>(Op);
11829
11830 EVT MemVT = VData.getValueType();
11831 return DAG.getMemIntrinsicNode(NewOpcode, DL, Op->getVTList(), Ops, MemVT,
11832 M->getMemOperand());
11833}
11834
11835SDValue
11836SITargetLowering::lowerStructBufferAtomicIntrin(SDValue Op, SelectionDAG &DAG,
11837 unsigned NewOpcode) const {
11838 SDLoc DL(Op);
11839
11840 SDValue VData = Op.getOperand(2);
11841 if (VData.getValueSizeInBits() != 32 && VData.getValueSizeInBits() != 64) {
11842 SmallVector<EVT, 2> ResultTypes(Op->values());
11843 return diagnoseUnsupportedImage(DAG, Op, ResultTypes, DL,
11844 "unsupported buffer atomic data type");
11845 }
11846 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
11847 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(5), DAG);
11848 auto SOffset = selectSOffset(Op.getOperand(6), DAG, Subtarget);
11849 SDValue Ops[] = {
11850 Op.getOperand(0), // Chain
11851 VData, // vdata
11852 Rsrc, // rsrc
11853 Op.getOperand(4), // vindex
11854 VOffset, // voffset
11855 SOffset, // soffset
11856 Offset, // offset
11857 Op.getOperand(7), // cachepolicy
11858 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
11859 };
11860
11861 auto *M = cast<MemSDNode>(Op);
11862
11863 EVT MemVT = VData.getValueType();
11864 return DAG.getMemIntrinsicNode(NewOpcode, DL, Op->getVTList(), Ops, MemVT,
11865 M->getMemOperand());
11866}
11867
11869 SDLoc DL) {
11870 SDNode *N = Op.getNode();
11871 SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
11872 unsigned NumOperands = N->getNumOperands();
11873 if (N->getOperand(NumOperands - 1) == Zero)
11874 return;
11876 Ops[NumOperands - 1] = Zero; // M0 = 0
11877 DAG.UpdateNodeOperands(N, Ops);
11878}
11879
11880SDValue SITargetLowering::LowerINTRINSIC_W_CHAIN(SDValue Op,
11881 SelectionDAG &DAG) const {
11882 unsigned IntrID = Op.getConstantOperandVal(1);
11883 SDLoc DL(Op);
11884
11885 switch (IntrID) {
11886 case Intrinsic::amdgcn_cluster_load_b32:
11887 case Intrinsic::amdgcn_cluster_load_b64:
11888 case Intrinsic::amdgcn_cluster_load_b128: {
11889 if (Subtarget->hasGFX1250_STRICT())
11891 return SDValue();
11892 }
11893 case Intrinsic::amdgcn_ds_ordered_add:
11894 case Intrinsic::amdgcn_ds_ordered_swap: {
11895 MemSDNode *M = cast<MemSDNode>(Op);
11896 SDValue Chain = M->getOperand(0);
11897 SDValue M0 = M->getOperand(2);
11898 SDValue Value = M->getOperand(3);
11899 unsigned IndexOperand = M->getConstantOperandVal(7);
11900 unsigned WaveRelease = M->getConstantOperandVal(8);
11901 unsigned WaveDone = M->getConstantOperandVal(9);
11902
11903 unsigned OrderedCountIndex = IndexOperand & 0x3f;
11904 IndexOperand &= ~0x3f;
11905 unsigned CountDw = 0;
11906
11907 if (Subtarget->getGeneration() >= AMDGPUSubtarget::GFX10) {
11908 CountDw = (IndexOperand >> 24) & 0xf;
11909 IndexOperand &= ~(0xf << 24);
11910
11911 if (CountDw < 1 || CountDw > 4) {
11912 const Function &Fn = DAG.getMachineFunction().getFunction();
11913 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
11914 Fn, "ds_ordered_count: dword count must be between 1 and 4",
11915 DL.getDebugLoc()));
11916 CountDw = 1;
11917 }
11918 }
11919
11920 if (IndexOperand) {
11921 const Function &Fn = DAG.getMachineFunction().getFunction();
11922 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
11923 Fn, "ds_ordered_count: bad index operand", DL.getDebugLoc()));
11924 }
11925
11926 if (WaveDone && !WaveRelease) {
11927 // TODO: Move this to IR verifier
11928 const Function &Fn = DAG.getMachineFunction().getFunction();
11929 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
11930 Fn, "ds_ordered_count: wave_done requires wave_release",
11931 DL.getDebugLoc()));
11932 }
11933
11934 unsigned Instruction = IntrID == Intrinsic::amdgcn_ds_ordered_add ? 0 : 1;
11935 unsigned ShaderType =
11937 unsigned Offset0 = OrderedCountIndex << 2;
11938 unsigned Offset1 = WaveRelease | (WaveDone << 1) | (Instruction << 4);
11939
11940 if (Subtarget->getGeneration() >= AMDGPUSubtarget::GFX10)
11941 Offset1 |= (CountDw - 1) << 6;
11942
11943 if (Subtarget->getGeneration() < AMDGPUSubtarget::GFX11)
11944 Offset1 |= ShaderType << 2;
11945
11946 unsigned Offset = Offset0 | (Offset1 << 8);
11947
11948 SDValue Ops[] = {
11949 Chain, Value, DAG.getTargetConstant(Offset, DL, MVT::i16),
11950 copyToM0(DAG, Chain, DL, M0).getValue(1), // Glue
11951 };
11952 return DAG.getMemIntrinsicNode(AMDGPUISD::DS_ORDERED_COUNT, DL,
11953 M->getVTList(), Ops, M->getMemoryVT(),
11954 M->getMemOperand());
11955 }
11956 case Intrinsic::amdgcn_ptr_s_buffer_load: {
11957 unsigned CPol = Op.getConstantOperandVal(4);
11958 if (CPol & ~((Subtarget->getGeneration() >= AMDGPUSubtarget::GFX12)
11961 return Op;
11962
11963 MemSDNode *M = cast<MemSDNode>(Op);
11964 return lowerSBuffer(
11965 Op.getValueType(), M->getMemoryVT(), DL, Op.getOperand(0),
11966 bufferRsrcPtrToVector(Op.getOperand(2), DAG), Op.getOperand(3),
11967 Op.getOperand(4), DAG, M->getMemOperand());
11968 }
11969 case Intrinsic::amdgcn_raw_buffer_load:
11970 case Intrinsic::amdgcn_raw_ptr_buffer_load:
11971 case Intrinsic::amdgcn_raw_atomic_buffer_load:
11972 case Intrinsic::amdgcn_raw_ptr_atomic_buffer_load:
11973 case Intrinsic::amdgcn_raw_buffer_load_format:
11974 case Intrinsic::amdgcn_raw_ptr_buffer_load_format: {
11975 const bool IsFormat =
11976 IntrID == Intrinsic::amdgcn_raw_buffer_load_format ||
11977 IntrID == Intrinsic::amdgcn_raw_ptr_buffer_load_format;
11978
11979 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(2), DAG);
11980 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(3), DAG);
11981 auto SOffset = selectSOffset(Op.getOperand(4), DAG, Subtarget);
11982 SDValue Ops[] = {
11983 Op.getOperand(0), // Chain
11984 Rsrc, // rsrc
11985 DAG.getConstant(0, DL, MVT::i32), // vindex
11986 VOffset, // voffset
11987 SOffset, // soffset
11988 Offset, // offset
11989 Op.getOperand(5), // cachepolicy, swizzled buffer
11990 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
11991 };
11992
11993 auto *M = cast<MemSDNode>(Op);
11994 return lowerIntrinsicLoad(M, IsFormat, DAG, Ops);
11995 }
11996 case Intrinsic::amdgcn_struct_buffer_load:
11997 case Intrinsic::amdgcn_struct_ptr_buffer_load:
11998 case Intrinsic::amdgcn_struct_buffer_load_format:
11999 case Intrinsic::amdgcn_struct_ptr_buffer_load_format:
12000 case Intrinsic::amdgcn_struct_atomic_buffer_load:
12001 case Intrinsic::amdgcn_struct_ptr_atomic_buffer_load: {
12002 const bool IsFormat =
12003 IntrID == Intrinsic::amdgcn_struct_buffer_load_format ||
12004 IntrID == Intrinsic::amdgcn_struct_ptr_buffer_load_format;
12005
12006 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(2), DAG);
12007 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(4), DAG);
12008 auto SOffset = selectSOffset(Op.getOperand(5), DAG, Subtarget);
12009 SDValue Ops[] = {
12010 Op.getOperand(0), // Chain
12011 Rsrc, // rsrc
12012 Op.getOperand(3), // vindex
12013 VOffset, // voffset
12014 SOffset, // soffset
12015 Offset, // offset
12016 Op.getOperand(6), // cachepolicy, swizzled buffer
12017 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
12018 };
12019
12020 return lowerIntrinsicLoad(cast<MemSDNode>(Op), IsFormat, DAG, Ops);
12021 }
12022 case Intrinsic::amdgcn_raw_tbuffer_load:
12023 case Intrinsic::amdgcn_raw_ptr_tbuffer_load: {
12024 MemSDNode *M = cast<MemSDNode>(Op);
12025 EVT LoadVT = Op.getValueType();
12026 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(2), DAG);
12027 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(3), DAG);
12028 auto SOffset = selectSOffset(Op.getOperand(4), DAG, Subtarget);
12029
12030 SDValue Ops[] = {
12031 Op.getOperand(0), // Chain
12032 Rsrc, // rsrc
12033 DAG.getConstant(0, DL, MVT::i32), // vindex
12034 VOffset, // voffset
12035 SOffset, // soffset
12036 Offset, // offset
12037 Op.getOperand(5), // format
12038 Op.getOperand(6), // cachepolicy, swizzled buffer
12039 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
12040 };
12041
12042 if (LoadVT.getScalarSizeInBits() == 16)
12043 return adjustLoadValueType(AMDGPUISD::TBUFFER_LOAD_FORMAT_D16, M, DAG,
12044 Ops);
12045 return getMemIntrinsicNode(AMDGPUISD::TBUFFER_LOAD_FORMAT, DL,
12046 Op->getVTList(), Ops, LoadVT, M->getMemOperand(),
12047 DAG);
12048 }
12049 case Intrinsic::amdgcn_struct_tbuffer_load:
12050 case Intrinsic::amdgcn_struct_ptr_tbuffer_load: {
12051 MemSDNode *M = cast<MemSDNode>(Op);
12052 EVT LoadVT = Op.getValueType();
12053 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(2), DAG);
12054 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(4), DAG);
12055 auto SOffset = selectSOffset(Op.getOperand(5), DAG, Subtarget);
12056
12057 SDValue Ops[] = {
12058 Op.getOperand(0), // Chain
12059 Rsrc, // rsrc
12060 Op.getOperand(3), // vindex
12061 VOffset, // voffset
12062 SOffset, // soffset
12063 Offset, // offset
12064 Op.getOperand(6), // format
12065 Op.getOperand(7), // cachepolicy, swizzled buffer
12066 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
12067 };
12068
12069 if (LoadVT.getScalarSizeInBits() == 16)
12070 return adjustLoadValueType(AMDGPUISD::TBUFFER_LOAD_FORMAT_D16, M, DAG,
12071 Ops);
12072 return getMemIntrinsicNode(AMDGPUISD::TBUFFER_LOAD_FORMAT, DL,
12073 Op->getVTList(), Ops, LoadVT, M->getMemOperand(),
12074 DAG);
12075 }
12076 case Intrinsic::amdgcn_raw_buffer_atomic_fadd:
12077 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fadd:
12078 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FADD);
12079 case Intrinsic::amdgcn_struct_buffer_atomic_fadd:
12080 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fadd:
12081 return lowerStructBufferAtomicIntrin(Op, DAG,
12082 AMDGPUISD::BUFFER_ATOMIC_FADD);
12083 case Intrinsic::amdgcn_raw_buffer_atomic_fmin:
12084 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmin:
12085 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FMIN);
12086 case Intrinsic::amdgcn_struct_buffer_atomic_fmin:
12087 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fmin:
12088 return lowerStructBufferAtomicIntrin(Op, DAG,
12089 AMDGPUISD::BUFFER_ATOMIC_FMIN);
12090 case Intrinsic::amdgcn_raw_buffer_atomic_fmax:
12091 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_fmax:
12092 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_FMAX);
12093 case Intrinsic::amdgcn_struct_buffer_atomic_fmax:
12094 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_fmax:
12095 return lowerStructBufferAtomicIntrin(Op, DAG,
12096 AMDGPUISD::BUFFER_ATOMIC_FMAX);
12097 case Intrinsic::amdgcn_raw_buffer_atomic_swap:
12098 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_swap:
12099 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SWAP);
12100 case Intrinsic::amdgcn_raw_buffer_atomic_add:
12101 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_add:
12102 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_ADD);
12103 case Intrinsic::amdgcn_raw_buffer_atomic_sub:
12104 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub:
12105 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SUB);
12106 case Intrinsic::amdgcn_raw_buffer_atomic_smin:
12107 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smin:
12108 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SMIN);
12109 case Intrinsic::amdgcn_raw_buffer_atomic_umin:
12110 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umin:
12111 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_UMIN);
12112 case Intrinsic::amdgcn_raw_buffer_atomic_smax:
12113 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_smax:
12114 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SMAX);
12115 case Intrinsic::amdgcn_raw_buffer_atomic_umax:
12116 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_umax:
12117 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_UMAX);
12118 case Intrinsic::amdgcn_raw_buffer_atomic_and:
12119 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_and:
12120 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_AND);
12121 case Intrinsic::amdgcn_raw_buffer_atomic_or:
12122 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_or:
12123 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_OR);
12124 case Intrinsic::amdgcn_raw_buffer_atomic_xor:
12125 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_xor:
12126 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_XOR);
12127 case Intrinsic::amdgcn_raw_buffer_atomic_inc:
12128 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_inc:
12129 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_INC);
12130 case Intrinsic::amdgcn_raw_buffer_atomic_dec:
12131 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_dec:
12132 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_DEC);
12133 case Intrinsic::amdgcn_struct_buffer_atomic_swap:
12134 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_swap:
12135 return lowerStructBufferAtomicIntrin(Op, DAG,
12136 AMDGPUISD::BUFFER_ATOMIC_SWAP);
12137 case Intrinsic::amdgcn_struct_buffer_atomic_add:
12138 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_add:
12139 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_ADD);
12140 case Intrinsic::amdgcn_struct_buffer_atomic_sub:
12141 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_sub:
12142 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_SUB);
12143 case Intrinsic::amdgcn_struct_buffer_atomic_smin:
12144 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smin:
12145 return lowerStructBufferAtomicIntrin(Op, DAG,
12146 AMDGPUISD::BUFFER_ATOMIC_SMIN);
12147 case Intrinsic::amdgcn_struct_buffer_atomic_umin:
12148 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umin:
12149 return lowerStructBufferAtomicIntrin(Op, DAG,
12150 AMDGPUISD::BUFFER_ATOMIC_UMIN);
12151 case Intrinsic::amdgcn_struct_buffer_atomic_smax:
12152 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_smax:
12153 return lowerStructBufferAtomicIntrin(Op, DAG,
12154 AMDGPUISD::BUFFER_ATOMIC_SMAX);
12155 case Intrinsic::amdgcn_struct_buffer_atomic_umax:
12156 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_umax:
12157 return lowerStructBufferAtomicIntrin(Op, DAG,
12158 AMDGPUISD::BUFFER_ATOMIC_UMAX);
12159 case Intrinsic::amdgcn_struct_buffer_atomic_and:
12160 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_and:
12161 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_AND);
12162 case Intrinsic::amdgcn_struct_buffer_atomic_or:
12163 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_or:
12164 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_OR);
12165 case Intrinsic::amdgcn_struct_buffer_atomic_xor:
12166 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_xor:
12167 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_XOR);
12168 case Intrinsic::amdgcn_struct_buffer_atomic_inc:
12169 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_inc:
12170 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_INC);
12171 case Intrinsic::amdgcn_struct_buffer_atomic_dec:
12172 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_dec:
12173 return lowerStructBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_DEC);
12174 case Intrinsic::amdgcn_raw_buffer_atomic_sub_clamp_u32:
12175 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_sub_clamp_u32:
12176 return lowerRawBufferAtomicIntrin(Op, DAG, AMDGPUISD::BUFFER_ATOMIC_CSUB);
12177 case Intrinsic::amdgcn_struct_buffer_atomic_sub_clamp_u32:
12178 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_sub_clamp_u32:
12179 return lowerStructBufferAtomicIntrin(Op, DAG,
12180 AMDGPUISD::BUFFER_ATOMIC_CSUB);
12181 case Intrinsic::amdgcn_raw_buffer_atomic_cond_sub_u32:
12182 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_cond_sub_u32:
12183 return lowerRawBufferAtomicIntrin(Op, DAG,
12184 AMDGPUISD::BUFFER_ATOMIC_COND_SUB_U32);
12185 case Intrinsic::amdgcn_struct_buffer_atomic_cond_sub_u32:
12186 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_cond_sub_u32:
12187 return lowerStructBufferAtomicIntrin(Op, DAG,
12188 AMDGPUISD::BUFFER_ATOMIC_COND_SUB_U32);
12189 case Intrinsic::amdgcn_raw_buffer_atomic_cmpswap:
12190 case Intrinsic::amdgcn_raw_ptr_buffer_atomic_cmpswap: {
12191 SDValue Src = Op.getOperand(2);
12192 if (Src.getValueSizeInBits() != 32 && Src.getValueSizeInBits() != 64) {
12193 SmallVector<EVT, 2> ResultTypes(Op->values());
12194 return diagnoseUnsupportedImage(DAG, Op, ResultTypes, DL,
12195 "unsupported buffer atomic data type");
12196 }
12197 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(4), DAG);
12198 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(5), DAG);
12199 auto SOffset = selectSOffset(Op.getOperand(6), DAG, Subtarget);
12200 SDValue Ops[] = {
12201 Op.getOperand(0), // Chain
12202 Op.getOperand(2), // src
12203 Op.getOperand(3), // cmp
12204 Rsrc, // rsrc
12205 DAG.getConstant(0, DL, MVT::i32), // vindex
12206 VOffset, // voffset
12207 SOffset, // soffset
12208 Offset, // offset
12209 Op.getOperand(7), // cachepolicy
12210 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
12211 };
12212 EVT VT = Op.getValueType();
12213 auto *M = cast<MemSDNode>(Op);
12214
12215 return DAG.getMemIntrinsicNode(AMDGPUISD::BUFFER_ATOMIC_CMPSWAP, DL,
12216 Op->getVTList(), Ops, VT,
12217 M->getMemOperand());
12218 }
12219 case Intrinsic::amdgcn_struct_buffer_atomic_cmpswap:
12220 case Intrinsic::amdgcn_struct_ptr_buffer_atomic_cmpswap: {
12221 SDValue Src = Op.getOperand(2);
12222 if (Src.getValueSizeInBits() != 32 && Src.getValueSizeInBits() != 64) {
12223 SmallVector<EVT, 2> ResultTypes(Op->values());
12224 return diagnoseUnsupportedImage(DAG, Op, ResultTypes, DL,
12225 "unsupported buffer atomic data type");
12226 }
12227 SDValue Rsrc = bufferRsrcPtrToVector(Op->getOperand(4), DAG);
12228 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(6), DAG);
12229 auto SOffset = selectSOffset(Op.getOperand(7), DAG, Subtarget);
12230 SDValue Ops[] = {
12231 Op.getOperand(0), // Chain
12232 Op.getOperand(2), // src
12233 Op.getOperand(3), // cmp
12234 Rsrc, // rsrc
12235 Op.getOperand(5), // vindex
12236 VOffset, // voffset
12237 SOffset, // soffset
12238 Offset, // offset
12239 Op.getOperand(8), // cachepolicy
12240 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
12241 };
12242 EVT VT = Op.getValueType();
12243 auto *M = cast<MemSDNode>(Op);
12244
12245 return DAG.getMemIntrinsicNode(AMDGPUISD::BUFFER_ATOMIC_CMPSWAP, DL,
12246 Op->getVTList(), Ops, VT,
12247 M->getMemOperand());
12248 }
12249 case Intrinsic::amdgcn_image_bvh_dual_intersect_ray:
12250 case Intrinsic::amdgcn_image_bvh8_intersect_ray: {
12251 MemSDNode *M = cast<MemSDNode>(Op);
12252 SDValue NodePtr = M->getOperand(2);
12253 SDValue RayExtent = M->getOperand(3);
12254 SDValue InstanceMask = M->getOperand(4);
12255 SDValue RayOrigin = M->getOperand(5);
12256 SDValue RayDir = M->getOperand(6);
12257 SDValue Offsets = M->getOperand(7);
12258 SDValue TDescr = M->getOperand(8);
12259
12260 assert(NodePtr.getValueType() == MVT::i64);
12261 assert(RayDir.getValueType() == MVT::v3f32);
12262
12263 bool IsBVH8 = IntrID == Intrinsic::amdgcn_image_bvh8_intersect_ray;
12264 const unsigned NumVDataDwords = 10;
12265 const unsigned NumVAddrDwords = IsBVH8 ? 11 : 12;
12266 int Opcode = AMDGPU::getMIMGOpcode(
12267 IsBVH8 ? AMDGPU::IMAGE_BVH8_INTERSECT_RAY
12268 : AMDGPU::IMAGE_BVH_DUAL_INTERSECT_RAY,
12269 AMDGPU::MIMGEncGfx12, NumVDataDwords, NumVAddrDwords);
12270 assert(Opcode != -1);
12271
12273 Ops.push_back(NodePtr);
12274 Ops.push_back(DAG.getBuildVector(
12275 MVT::v2i32, DL,
12276 {DAG.getBitcast(MVT::i32, RayExtent),
12277 DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, InstanceMask)}));
12278 Ops.push_back(RayOrigin);
12279 Ops.push_back(RayDir);
12280 Ops.push_back(Offsets);
12281 Ops.push_back(TDescr);
12282 Ops.push_back(M->getChain());
12283
12284 auto *NewNode = DAG.getMachineNode(Opcode, DL, M->getVTList(), Ops);
12285 MachineMemOperand *MemRef = M->getMemOperand();
12286 DAG.setNodeMemRefs(NewNode, {MemRef});
12287 return SDValue(NewNode, 0);
12288 }
12289 case Intrinsic::amdgcn_image_bvh_intersect_ray: {
12290 MemSDNode *M = cast<MemSDNode>(Op);
12291 SDValue NodePtr = M->getOperand(2);
12292 SDValue RayExtent = M->getOperand(3);
12293 SDValue RayOrigin = M->getOperand(4);
12294 SDValue RayDir = M->getOperand(5);
12295 SDValue RayInvDir = M->getOperand(6);
12296 SDValue TDescr = M->getOperand(7);
12297
12298 assert(NodePtr.getValueType() == MVT::i32 ||
12299 NodePtr.getValueType() == MVT::i64);
12300 assert(RayDir.getValueType() == MVT::v3f16 ||
12301 RayDir.getValueType() == MVT::v3f32);
12302
12303 const bool IsGFX11 = AMDGPU::isGFX11(*Subtarget);
12304 const bool IsGFX11Plus = AMDGPU::isGFX11Plus(*Subtarget);
12305 const bool IsGFX12Plus = AMDGPU::isGFX12Plus(*Subtarget);
12306 const bool IsA16 = RayDir.getValueType().getVectorElementType() == MVT::f16;
12307 const bool Is64 = NodePtr.getValueType() == MVT::i64;
12308 const unsigned NumVDataDwords = 4;
12309 const unsigned NumVAddrDwords = IsA16 ? (Is64 ? 9 : 8) : (Is64 ? 12 : 11);
12310 const unsigned NumVAddrs = IsGFX11Plus ? (IsA16 ? 4 : 5) : NumVAddrDwords;
12311 const bool UseNSA = (Subtarget->hasNSAEncoding() &&
12312 NumVAddrs <= Subtarget->getNSAMaxSize()) ||
12313 IsGFX12Plus;
12314 const unsigned BaseOpcodes[2][2] = {
12315 {AMDGPU::IMAGE_BVH_INTERSECT_RAY, AMDGPU::IMAGE_BVH_INTERSECT_RAY_a16},
12316 {AMDGPU::IMAGE_BVH64_INTERSECT_RAY,
12317 AMDGPU::IMAGE_BVH64_INTERSECT_RAY_a16}};
12318 int Opcode;
12319 if (UseNSA) {
12320 Opcode = AMDGPU::getMIMGOpcode(BaseOpcodes[Is64][IsA16],
12321 IsGFX12Plus ? AMDGPU::MIMGEncGfx12
12322 : IsGFX11 ? AMDGPU::MIMGEncGfx11NSA
12323 : AMDGPU::MIMGEncGfx10NSA,
12324 NumVDataDwords, NumVAddrDwords);
12325 } else {
12326 assert(!IsGFX12Plus);
12327 Opcode = AMDGPU::getMIMGOpcode(BaseOpcodes[Is64][IsA16],
12328 IsGFX11 ? AMDGPU::MIMGEncGfx11Default
12329 : AMDGPU::MIMGEncGfx10Default,
12330 NumVDataDwords, NumVAddrDwords);
12331 }
12332 assert(Opcode != -1);
12333
12335
12336 auto packLanes = [&DAG, &Ops, &DL](SDValue Op, bool IsAligned) {
12338 DAG.ExtractVectorElements(Op, Lanes, 0, 3);
12339 if (Lanes[0].getValueSizeInBits() == 32) {
12340 for (unsigned I = 0; I < 3; ++I)
12341 Ops.push_back(DAG.getBitcast(MVT::i32, Lanes[I]));
12342 } else {
12343 if (IsAligned) {
12344 Ops.push_back(DAG.getBitcast(
12345 MVT::i32,
12346 DAG.getBuildVector(MVT::v2f16, DL, {Lanes[0], Lanes[1]})));
12347 Ops.push_back(Lanes[2]);
12348 } else {
12349 SDValue Elt0 = Ops.pop_back_val();
12350 Ops.push_back(DAG.getBitcast(
12351 MVT::i32, DAG.getBuildVector(MVT::v2f16, DL, {Elt0, Lanes[0]})));
12352 Ops.push_back(DAG.getBitcast(
12353 MVT::i32,
12354 DAG.getBuildVector(MVT::v2f16, DL, {Lanes[1], Lanes[2]})));
12355 }
12356 }
12357 };
12358
12359 if (UseNSA && IsGFX11Plus) {
12360 Ops.push_back(NodePtr);
12361 Ops.push_back(DAG.getBitcast(MVT::i32, RayExtent));
12362 Ops.push_back(RayOrigin);
12363 if (IsA16) {
12364 SmallVector<SDValue, 3> DirLanes, InvDirLanes, MergedLanes;
12365 DAG.ExtractVectorElements(RayDir, DirLanes, 0, 3);
12366 DAG.ExtractVectorElements(RayInvDir, InvDirLanes, 0, 3);
12367 for (unsigned I = 0; I < 3; ++I) {
12368 MergedLanes.push_back(DAG.getBitcast(
12369 MVT::i32, DAG.getBuildVector(MVT::v2f16, DL,
12370 {DirLanes[I], InvDirLanes[I]})));
12371 }
12372 Ops.push_back(DAG.getBuildVector(MVT::v3i32, DL, MergedLanes));
12373 } else {
12374 Ops.push_back(RayDir);
12375 Ops.push_back(RayInvDir);
12376 }
12377 } else {
12378 if (Is64)
12379 DAG.ExtractVectorElements(DAG.getBitcast(MVT::v2i32, NodePtr), Ops, 0,
12380 2);
12381 else
12382 Ops.push_back(NodePtr);
12383
12384 Ops.push_back(DAG.getBitcast(MVT::i32, RayExtent));
12385 packLanes(RayOrigin, true);
12386 packLanes(RayDir, true);
12387 packLanes(RayInvDir, false);
12388 }
12389
12390 if (!UseNSA) {
12391 // Build a single vector containing all the operands so far prepared.
12392 if (NumVAddrDwords > 12) {
12393 SDValue Undef = DAG.getPOISON(MVT::i32);
12394 Ops.append(16 - Ops.size(), Undef);
12395 }
12396 assert(Ops.size() >= 8 && Ops.size() <= 12);
12397 SDValue MergedOps =
12398 DAG.getBuildVector(MVT::getVectorVT(MVT::i32, Ops.size()), DL, Ops);
12399 Ops.clear();
12400 Ops.push_back(MergedOps);
12401 }
12402
12403 Ops.push_back(TDescr);
12404 Ops.push_back(DAG.getTargetConstant(IsA16, DL, MVT::i1));
12405 Ops.push_back(M->getChain());
12406
12407 auto *NewNode = DAG.getMachineNode(Opcode, DL, M->getVTList(), Ops);
12408 MachineMemOperand *MemRef = M->getMemOperand();
12409 DAG.setNodeMemRefs(NewNode, {MemRef});
12410 return SDValue(NewNode, 0);
12411 }
12412 case Intrinsic::amdgcn_global_atomic_fmin_num:
12413 case Intrinsic::amdgcn_global_atomic_fmax_num:
12414 case Intrinsic::amdgcn_flat_atomic_fmin_num:
12415 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
12416 MemSDNode *M = cast<MemSDNode>(Op);
12417 SDValue Ops[] = {
12418 M->getOperand(0), // Chain
12419 M->getOperand(2), // Ptr
12420 M->getOperand(3) // Value
12421 };
12422 unsigned Opcode = 0;
12423 switch (IntrID) {
12424 case Intrinsic::amdgcn_global_atomic_fmin_num:
12425 case Intrinsic::amdgcn_flat_atomic_fmin_num: {
12426 Opcode = ISD::ATOMIC_LOAD_FMIN;
12427 break;
12428 }
12429 case Intrinsic::amdgcn_global_atomic_fmax_num:
12430 case Intrinsic::amdgcn_flat_atomic_fmax_num: {
12431 Opcode = ISD::ATOMIC_LOAD_FMAX;
12432 break;
12433 }
12434 default:
12435 llvm_unreachable("unhandled atomic opcode");
12436 }
12437 return DAG.getAtomic(Opcode, SDLoc(Op), M->getMemoryVT(), M->getVTList(),
12438 Ops, M->getMemOperand());
12439 }
12440 case Intrinsic::amdgcn_s_alloc_vgpr: {
12441 SDValue NumVGPRs = Op.getOperand(2);
12442 if (!NumVGPRs->isDivergent())
12443 return Op;
12444
12445 SDValue ReadFirstLaneID =
12446 DAG.getTargetConstant(Intrinsic::amdgcn_readfirstlane, DL, MVT::i32);
12447 NumVGPRs = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, MVT::i32,
12448 ReadFirstLaneID, NumVGPRs);
12449
12450 return DAG.getNode(ISD::INTRINSIC_W_CHAIN, DL, Op->getVTList(),
12451 Op.getOperand(0), Op.getOperand(1), NumVGPRs);
12452 }
12453 case Intrinsic::amdgcn_s_get_barrier_state:
12454 case Intrinsic::amdgcn_s_get_named_barrier_state: {
12455 SDValue Chain = Op->getOperand(0);
12457 unsigned Opc;
12458
12459 if (isa<ConstantSDNode>(Op->getOperand(2))) {
12460 uint64_t BarID = cast<ConstantSDNode>(Op->getOperand(2))->getZExtValue();
12461 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state)
12462 BarID = BarID & 0x3F;
12463 Opc = AMDGPU::S_GET_BARRIER_STATE_IMM;
12464 SDValue K = DAG.getTargetConstant(BarID, DL, MVT::i32);
12465 Ops.push_back(K);
12466 Ops.push_back(Chain);
12467 } else {
12468 Opc = AMDGPU::S_GET_BARRIER_STATE_M0;
12469 if (IntrID == Intrinsic::amdgcn_s_get_named_barrier_state) {
12470 SDValue M0Val = DAG.getNode(ISD::AND, DL, MVT::i32, Op->getOperand(2),
12471 DAG.getConstant(0x3F, DL, MVT::i32));
12472 Ops.push_back(copyToM0(DAG, Chain, DL, M0Val).getValue(0));
12473 } else
12474 Ops.push_back(copyToM0(DAG, Chain, DL, Op->getOperand(2)).getValue(0));
12475 }
12476
12477 auto *NewMI = DAG.getMachineNode(Opc, DL, Op->getVTList(), Ops);
12478 return SDValue(NewMI, 0);
12479 }
12480 case Intrinsic::amdgcn_cooperative_atomic_load_32x4B:
12481 case Intrinsic::amdgcn_cooperative_atomic_load_16x8B:
12482 case Intrinsic::amdgcn_cooperative_atomic_load_8x16B: {
12483 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
12484 SDValue Chain = Op->getOperand(0);
12485 SDValue Ptr = Op->getOperand(2);
12486 EVT VT = Op->getValueType(0);
12487 return DAG.getAtomicLoad(ISD::NON_EXTLOAD, DL, MII->getMemoryVT(), VT,
12488 Chain, Ptr, MII->getMemOperand());
12489 }
12490 case Intrinsic::amdgcn_av_load_b128: {
12491 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
12492 SDValue Chain = Op->getOperand(0);
12493 SDValue Ptr = Op->getOperand(2);
12494 EVT VT = Op->getValueType(0);
12495 // Lower to a regular ISD::LOAD. The MachineMemOperand carries Monotonic
12496 // ordering and syncscope so that SIMemoryLegalizer sets cache policy bits.
12497 // Address space filtering in the load_global/load_flat PatFrags selects
12498 // the correct GLOBAL vs FLAT instruction.
12499 return DAG.getLoad(VT, DL, Chain, Ptr, MII->getMemOperand());
12500 }
12501 case Intrinsic::amdgcn_flat_load_monitor_b32:
12502 case Intrinsic::amdgcn_flat_load_monitor_b64:
12503 case Intrinsic::amdgcn_flat_load_monitor_b128: {
12504 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
12505 SDValue Chain = Op->getOperand(0);
12506 SDValue Ptr = Op->getOperand(2);
12507 return DAG.getMemIntrinsicNode(AMDGPUISD::FLAT_LOAD_MONITOR, DL,
12508 Op->getVTList(), {Chain, Ptr},
12509 MII->getMemoryVT(), MII->getMemOperand());
12510 }
12511 case Intrinsic::amdgcn_global_load_monitor_b32:
12512 case Intrinsic::amdgcn_global_load_monitor_b64:
12513 case Intrinsic::amdgcn_global_load_monitor_b128: {
12514 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
12515 SDValue Chain = Op->getOperand(0);
12516 SDValue Ptr = Op->getOperand(2);
12517 return DAG.getMemIntrinsicNode(AMDGPUISD::GLOBAL_LOAD_MONITOR, DL,
12518 Op->getVTList(), {Chain, Ptr},
12519 MII->getMemoryVT(), MII->getMemOperand());
12520 }
12521 default:
12522
12523 if (const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
12525 return lowerImage(Op, ImageDimIntr, DAG, true);
12526
12527 return SDValue();
12528 }
12529}
12530
12531// Call DAG.getMemIntrinsicNode for a load, but first widen a dwordx3 type to
12532// dwordx4 if on SI and handle TFE loads.
12533SDValue SITargetLowering::getMemIntrinsicNode(unsigned Opcode, const SDLoc &DL,
12534 SDVTList VTList,
12535 ArrayRef<SDValue> Ops, EVT MemVT,
12536 MachineMemOperand *MMO,
12537 SelectionDAG &DAG) const {
12538 LLVMContext &C = *DAG.getContext();
12540 EVT VT = VTList.VTs[0];
12541
12542 assert(VTList.NumVTs == 2 || VTList.NumVTs == 3);
12543 bool IsTFE = VTList.NumVTs == 3;
12544 if (IsTFE) {
12545 unsigned NumValueDWords = divideCeil(VT.getSizeInBits(), 32);
12546 unsigned NumOpDWords = NumValueDWords + 1;
12547 EVT OpDWordsVT = EVT::getVectorVT(C, MVT::i32, NumOpDWords);
12548 SDVTList OpDWordsVTList = DAG.getVTList(OpDWordsVT, VTList.VTs[2]);
12549 MachineMemOperand *OpDWordsMMO =
12550 MF.getMachineMemOperand(MMO, 0, NumOpDWords * 4);
12551 SDValue Op = getMemIntrinsicNode(Opcode, DL, OpDWordsVTList, Ops,
12552 OpDWordsVT, OpDWordsMMO, DAG);
12553 auto [Value, Status] = splitTFEValueAndStatus(Op, VT, DL, DAG);
12554 return DAG.getMergeValues({Value, Status, SDValue(Op.getNode(), 1)}, DL);
12555 }
12556
12557 if (!Subtarget->hasDwordx3LoadStores() &&
12558 (VT == MVT::v3i32 || VT == MVT::v3f32)) {
12559 EVT WidenedVT = EVT::getVectorVT(C, VT.getVectorElementType(), 4);
12560 EVT WidenedMemVT = EVT::getVectorVT(C, MemVT.getVectorElementType(), 4);
12561 MachineMemOperand *WidenedMMO = MF.getMachineMemOperand(MMO, 0, 16);
12562 SDVTList WidenedVTList = DAG.getVTList(WidenedVT, VTList.VTs[1]);
12563 SDValue Op = DAG.getMemIntrinsicNode(Opcode, DL, WidenedVTList, Ops,
12564 WidenedMemVT, WidenedMMO);
12565 SDValue Value = DAG.getNode(ISD::EXTRACT_SUBVECTOR, DL, VT, Op,
12566 DAG.getVectorIdxConstant(0, DL));
12567 return DAG.getMergeValues({Value, SDValue(Op.getNode(), 1)}, DL);
12568 }
12569
12570 return DAG.getMemIntrinsicNode(Opcode, DL, VTList, Ops, MemVT, MMO);
12571}
12572
12573SDValue SITargetLowering::handleD16VData(SDValue VData, SelectionDAG &DAG,
12574 bool ImageStore) const {
12575 EVT StoreVT = VData.getValueType();
12576
12577 // No change for f16 and legal vector D16 types.
12578 if (!StoreVT.isVector())
12579 return VData;
12580
12581 SDLoc DL(VData);
12582 unsigned NumElements = StoreVT.getVectorNumElements();
12583
12584 if (Subtarget->hasUnpackedD16VMem()) {
12585 // We need to unpack the packed data to store.
12586 EVT IntStoreVT = StoreVT.changeTypeToInteger();
12587 SDValue IntVData = DAG.getNode(ISD::BITCAST, DL, IntStoreVT, VData);
12588
12589 EVT EquivStoreVT =
12590 EVT::getVectorVT(*DAG.getContext(), MVT::i32, NumElements);
12591 SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, EquivStoreVT, IntVData);
12592 return DAG.UnrollVectorOp(ZExt.getNode());
12593 }
12594
12595 // The sq block of gfx8.1 does not estimate register use correctly for d16
12596 // image store instructions. The data operand is computed as if it were not a
12597 // d16 image instruction.
12598 if (ImageStore && Subtarget->hasImageStoreD16Bug()) {
12599 // Bitcast to i16
12600 EVT IntStoreVT = StoreVT.changeTypeToInteger();
12601 SDValue IntVData = DAG.getNode(ISD::BITCAST, DL, IntStoreVT, VData);
12602
12603 // Decompose into scalars
12605 DAG.ExtractVectorElements(IntVData, Elts);
12606
12607 // Group pairs of i16 into v2i16 and bitcast to i32
12608 SmallVector<SDValue, 4> PackedElts;
12609 for (unsigned I = 0; I < Elts.size() / 2; I += 1) {
12610 SDValue Pair =
12611 DAG.getBuildVector(MVT::v2i16, DL, {Elts[I * 2], Elts[I * 2 + 1]});
12612 SDValue IntPair = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Pair);
12613 PackedElts.push_back(IntPair);
12614 }
12615 if ((NumElements % 2) == 1) {
12616 // Handle v3i16
12617 unsigned I = Elts.size() / 2;
12618 SDValue Pair = DAG.getBuildVector(MVT::v2i16, DL,
12619 {Elts[I * 2], DAG.getPOISON(MVT::i16)});
12620 SDValue IntPair = DAG.getNode(ISD::BITCAST, DL, MVT::i32, Pair);
12621 PackedElts.push_back(IntPair);
12622 }
12623
12624 // Pad using UNDEF
12625 PackedElts.resize(Elts.size(), DAG.getPOISON(MVT::i32));
12626
12627 // Build final vector
12628 EVT VecVT =
12629 EVT::getVectorVT(*DAG.getContext(), MVT::i32, PackedElts.size());
12630 return DAG.getBuildVector(VecVT, DL, PackedElts);
12631 }
12632
12633 if (NumElements == 3) {
12634 EVT IntStoreVT =
12636 SDValue IntVData = DAG.getNode(ISD::BITCAST, DL, IntStoreVT, VData);
12637
12638 EVT WidenedStoreVT = EVT::getVectorVT(
12639 *DAG.getContext(), StoreVT.getVectorElementType(), NumElements + 1);
12640 EVT WidenedIntVT = EVT::getIntegerVT(*DAG.getContext(),
12641 WidenedStoreVT.getStoreSizeInBits());
12642 SDValue ZExt = DAG.getNode(ISD::ZERO_EXTEND, DL, WidenedIntVT, IntVData);
12643 return DAG.getNode(ISD::BITCAST, DL, WidenedStoreVT, ZExt);
12644 }
12645
12646 assert(isTypeLegal(StoreVT));
12647 return VData;
12648}
12649
12650static bool isAsyncLDSDMA(Intrinsic::ID Intr) {
12651 switch (Intr) {
12652 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
12653 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
12654 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
12655 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds:
12656 case Intrinsic::amdgcn_load_async_to_lds:
12657 case Intrinsic::amdgcn_global_load_async_lds:
12658 return true;
12659 }
12660 return false;
12661}
12662
12663SDValue SITargetLowering::LowerINTRINSIC_VOID(SDValue Op,
12664 SelectionDAG &DAG) const {
12665 SDLoc DL(Op);
12666 SDValue Chain = Op.getOperand(0);
12667 unsigned IntrinsicID = Op.getConstantOperandVal(1);
12668
12669 switch (IntrinsicID) {
12670 case Intrinsic::amdgcn_cluster_load_async_to_lds_b8:
12671 case Intrinsic::amdgcn_cluster_load_async_to_lds_b32:
12672 case Intrinsic::amdgcn_cluster_load_async_to_lds_b64:
12673 case Intrinsic::amdgcn_cluster_load_async_to_lds_b128: {
12674 if (Subtarget->hasGFX1250_STRICT())
12676 return SDValue();
12677 }
12678 case Intrinsic::amdgcn_exp_compr: {
12679 SDValue Src0 = Op.getOperand(4);
12680 SDValue Src1 = Op.getOperand(5);
12681 // Hack around illegal type on SI by directly selecting it.
12682 if (isTypeLegal(Src0.getValueType()))
12683 return SDValue();
12684
12685 const ConstantSDNode *Done = cast<ConstantSDNode>(Op.getOperand(6));
12686 SDValue Undef = DAG.getPOISON(MVT::f32);
12687 const SDValue Ops[] = {
12688 Op.getOperand(2), // tgt
12689 DAG.getNode(ISD::BITCAST, DL, MVT::f32, Src0), // src0
12690 DAG.getNode(ISD::BITCAST, DL, MVT::f32, Src1), // src1
12691 Undef, // src2
12692 Undef, // src3
12693 Op.getOperand(7), // vm
12694 DAG.getTargetConstant(1, DL, MVT::i1), // compr
12695 Op.getOperand(3), // en
12696 Op.getOperand(0) // Chain
12697 };
12698
12699 unsigned Opc = Done->isZero() ? AMDGPU::EXP : AMDGPU::EXP_DONE;
12700 return SDValue(DAG.getMachineNode(Opc, DL, Op->getVTList(), Ops), 0);
12701 }
12702
12703 case Intrinsic::amdgcn_struct_tbuffer_store:
12704 case Intrinsic::amdgcn_struct_ptr_tbuffer_store: {
12705 SDValue VData = Op.getOperand(2);
12706 bool IsD16 = (VData.getValueType().getScalarSizeInBits() == 16);
12707 if (IsD16)
12708 VData = handleD16VData(VData, DAG);
12709 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
12710 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(5), DAG);
12711 auto SOffset = selectSOffset(Op.getOperand(6), DAG, Subtarget);
12712 SDValue Ops[] = {
12713 Chain,
12714 VData, // vdata
12715 Rsrc, // rsrc
12716 Op.getOperand(4), // vindex
12717 VOffset, // voffset
12718 SOffset, // soffset
12719 Offset, // offset
12720 Op.getOperand(7), // format
12721 Op.getOperand(8), // cachepolicy, swizzled buffer
12722 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
12723 };
12724 unsigned Opc = IsD16 ? AMDGPUISD::TBUFFER_STORE_FORMAT_D16
12725 : AMDGPUISD::TBUFFER_STORE_FORMAT;
12726 MemSDNode *M = cast<MemSDNode>(Op);
12727 return DAG.getMemIntrinsicNode(Opc, DL, Op->getVTList(), Ops,
12728 M->getMemoryVT(), M->getMemOperand());
12729 }
12730
12731 case Intrinsic::amdgcn_raw_tbuffer_store:
12732 case Intrinsic::amdgcn_raw_ptr_tbuffer_store: {
12733 SDValue VData = Op.getOperand(2);
12734 bool IsD16 = (VData.getValueType().getScalarSizeInBits() == 16);
12735 if (IsD16)
12736 VData = handleD16VData(VData, DAG);
12737 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
12738 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(4), DAG);
12739 auto SOffset = selectSOffset(Op.getOperand(5), DAG, Subtarget);
12740 SDValue Ops[] = {
12741 Chain,
12742 VData, // vdata
12743 Rsrc, // rsrc
12744 DAG.getConstant(0, DL, MVT::i32), // vindex
12745 VOffset, // voffset
12746 SOffset, // soffset
12747 Offset, // offset
12748 Op.getOperand(6), // format
12749 Op.getOperand(7), // cachepolicy, swizzled buffer
12750 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
12751 };
12752 unsigned Opc = IsD16 ? AMDGPUISD::TBUFFER_STORE_FORMAT_D16
12753 : AMDGPUISD::TBUFFER_STORE_FORMAT;
12754 MemSDNode *M = cast<MemSDNode>(Op);
12755 return DAG.getMemIntrinsicNode(Opc, DL, Op->getVTList(), Ops,
12756 M->getMemoryVT(), M->getMemOperand());
12757 }
12758
12759 case Intrinsic::amdgcn_raw_buffer_store:
12760 case Intrinsic::amdgcn_raw_ptr_buffer_store:
12761 case Intrinsic::amdgcn_raw_buffer_store_format:
12762 case Intrinsic::amdgcn_raw_ptr_buffer_store_format: {
12763 const bool IsFormat =
12764 IntrinsicID == Intrinsic::amdgcn_raw_buffer_store_format ||
12765 IntrinsicID == Intrinsic::amdgcn_raw_ptr_buffer_store_format;
12766
12767 SDValue VData = Op.getOperand(2);
12768 EVT VDataVT = VData.getValueType();
12769 EVT EltType = VDataVT.getScalarType();
12770 bool IsD16 = IsFormat && (EltType.getSizeInBits() == 16);
12771
12772 if (IsFormat && !IsD16 && EltType.getSizeInBits() < 32) {
12773 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
12775 "unsupported sub-dword format buffer store", DL.getDebugLoc()));
12776 return Chain;
12777 }
12778
12779 if (IsD16) {
12780 VData = handleD16VData(VData, DAG);
12781 VDataVT = VData.getValueType();
12782 }
12783
12784 if (!isTypeLegal(VDataVT)) {
12785 VData =
12786 DAG.getNode(ISD::BITCAST, DL,
12787 getEquivalentMemType(*DAG.getContext(), VDataVT), VData);
12788 }
12789
12790 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
12791 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(4), DAG);
12792 auto SOffset = selectSOffset(Op.getOperand(5), DAG, Subtarget);
12793 SDValue Ops[] = {
12794 Chain,
12795 VData,
12796 Rsrc,
12797 DAG.getConstant(0, DL, MVT::i32), // vindex
12798 VOffset, // voffset
12799 SOffset, // soffset
12800 Offset, // offset
12801 Op.getOperand(6), // cachepolicy, swizzled buffer
12802 DAG.getTargetConstant(0, DL, MVT::i1), // idxen
12803 };
12804 unsigned Opc =
12805 IsFormat ? AMDGPUISD::BUFFER_STORE_FORMAT : AMDGPUISD::BUFFER_STORE;
12806 Opc = IsD16 ? AMDGPUISD::BUFFER_STORE_FORMAT_D16 : Opc;
12807 MemSDNode *M = cast<MemSDNode>(Op);
12808
12809 // Handle BUFFER_STORE_BYTE/SHORT overloaded intrinsics
12810 if (!IsD16 && !VDataVT.isVector() && EltType.getSizeInBits() < 32)
12811 return handleByteShortBufferStores(DAG, VDataVT, DL, Ops, M);
12812
12813 return DAG.getMemIntrinsicNode(Opc, DL, Op->getVTList(), Ops,
12814 M->getMemoryVT(), M->getMemOperand());
12815 }
12816
12817 case Intrinsic::amdgcn_struct_buffer_store:
12818 case Intrinsic::amdgcn_struct_ptr_buffer_store:
12819 case Intrinsic::amdgcn_struct_buffer_store_format:
12820 case Intrinsic::amdgcn_struct_ptr_buffer_store_format: {
12821 const bool IsFormat =
12822 IntrinsicID == Intrinsic::amdgcn_struct_buffer_store_format ||
12823 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_store_format;
12824
12825 SDValue VData = Op.getOperand(2);
12826 EVT VDataVT = VData.getValueType();
12827 EVT EltType = VDataVT.getScalarType();
12828 bool IsD16 = IsFormat && (EltType.getSizeInBits() == 16);
12829
12830 if (IsFormat && !IsD16 && EltType.getSizeInBits() < 32) {
12831 DAG.getContext()->diagnose(DiagnosticInfoUnsupported(
12833 "unsupported sub-dword format buffer store", DL.getDebugLoc()));
12834 return Chain;
12835 }
12836
12837 if (IsD16) {
12838 VData = handleD16VData(VData, DAG);
12839 VDataVT = VData.getValueType();
12840 }
12841
12842 if (!isTypeLegal(VDataVT)) {
12843 VData =
12844 DAG.getNode(ISD::BITCAST, DL,
12845 getEquivalentMemType(*DAG.getContext(), VDataVT), VData);
12846 }
12847
12848 auto Rsrc = bufferRsrcPtrToVector(Op.getOperand(3), DAG);
12849 auto [VOffset, Offset] = splitBufferOffsets(Op.getOperand(5), DAG);
12850 auto SOffset = selectSOffset(Op.getOperand(6), DAG, Subtarget);
12851 SDValue Ops[] = {
12852 Chain,
12853 VData,
12854 Rsrc,
12855 Op.getOperand(4), // vindex
12856 VOffset, // voffset
12857 SOffset, // soffset
12858 Offset, // offset
12859 Op.getOperand(7), // cachepolicy, swizzled buffer
12860 DAG.getTargetConstant(1, DL, MVT::i1), // idxen
12861 };
12862 unsigned Opc =
12863 !IsFormat ? AMDGPUISD::BUFFER_STORE : AMDGPUISD::BUFFER_STORE_FORMAT;
12864 Opc = IsD16 ? AMDGPUISD::BUFFER_STORE_FORMAT_D16 : Opc;
12865 MemSDNode *M = cast<MemSDNode>(Op);
12866
12867 // Handle BUFFER_STORE_BYTE/SHORT overloaded intrinsics
12868 EVT VDataType = VData.getValueType().getScalarType();
12869 if (!IsD16 && !VDataVT.isVector() && EltType.getSizeInBits() < 32)
12870 return handleByteShortBufferStores(DAG, VDataType, DL, Ops, M);
12871
12872 return DAG.getMemIntrinsicNode(Opc, DL, Op->getVTList(), Ops,
12873 M->getMemoryVT(), M->getMemOperand());
12874 }
12875 case Intrinsic::amdgcn_raw_buffer_load_lds:
12876 case Intrinsic::amdgcn_raw_buffer_load_async_lds:
12877 case Intrinsic::amdgcn_raw_ptr_buffer_load_lds:
12878 case Intrinsic::amdgcn_raw_ptr_buffer_load_async_lds:
12879 case Intrinsic::amdgcn_struct_buffer_load_lds:
12880 case Intrinsic::amdgcn_struct_buffer_load_async_lds:
12881 case Intrinsic::amdgcn_struct_ptr_buffer_load_lds:
12882 case Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds: {
12883 unsigned Opc;
12884 bool HasVIndex =
12885 IntrinsicID == Intrinsic::amdgcn_struct_buffer_load_lds ||
12886 IntrinsicID == Intrinsic::amdgcn_struct_buffer_load_async_lds ||
12887 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_load_lds ||
12888 IntrinsicID == Intrinsic::amdgcn_struct_ptr_buffer_load_async_lds;
12889 unsigned OpOffset = HasVIndex ? 1 : 0;
12890 SDValue VOffset = Op.getOperand(5 + OpOffset);
12891 bool HasVOffset = !isNullConstant(VOffset);
12892 unsigned Size = Op->getConstantOperandVal(4);
12893
12894 switch (Size) {
12895 default:
12896 return SDValue();
12897 case 1:
12898 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_BOTHEN
12899 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_IDXEN
12900 : HasVOffset ? AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFEN
12901 : AMDGPU::BUFFER_LOAD_UBYTE_LDS_OFFSET;
12902 break;
12903 case 2:
12904 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_BOTHEN
12905 : AMDGPU::BUFFER_LOAD_USHORT_LDS_IDXEN
12906 : HasVOffset ? AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFEN
12907 : AMDGPU::BUFFER_LOAD_USHORT_LDS_OFFSET;
12908 break;
12909 case 4:
12910 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_BOTHEN
12911 : AMDGPU::BUFFER_LOAD_DWORD_LDS_IDXEN
12912 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFEN
12913 : AMDGPU::BUFFER_LOAD_DWORD_LDS_OFFSET;
12914 break;
12915 case 12:
12916 if (!Subtarget->hasLDSLoadB96_B128())
12917 return SDValue();
12918 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_BOTHEN
12919 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_IDXEN
12920 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFEN
12921 : AMDGPU::BUFFER_LOAD_DWORDX3_LDS_OFFSET;
12922 break;
12923 case 16:
12924 if (!Subtarget->hasLDSLoadB96_B128())
12925 return SDValue();
12926 Opc = HasVIndex ? HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_BOTHEN
12927 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_IDXEN
12928 : HasVOffset ? AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFEN
12929 : AMDGPU::BUFFER_LOAD_DWORDX4_LDS_OFFSET;
12930 break;
12931 }
12932
12933 SDValue M0Val = copyToM0(DAG, Chain, DL, Op.getOperand(3));
12934
12936
12937 if (HasVIndex && HasVOffset)
12938 Ops.push_back(DAG.getBuildVector(MVT::v2i32, DL,
12939 {Op.getOperand(5), // VIndex
12940 VOffset}));
12941 else if (HasVIndex)
12942 Ops.push_back(Op.getOperand(5));
12943 else if (HasVOffset)
12944 Ops.push_back(VOffset);
12945
12946 SDValue Rsrc = bufferRsrcPtrToVector(Op.getOperand(2), DAG);
12947 Ops.push_back(Rsrc);
12948 Ops.push_back(Op.getOperand(6 + OpOffset)); // soffset
12949 Ops.push_back(Op.getOperand(7 + OpOffset)); // imm offset
12950 bool IsGFX12Plus = AMDGPU::isGFX12Plus(*Subtarget);
12951 unsigned Aux = Op.getConstantOperandVal(8 + OpOffset);
12952 Ops.push_back(DAG.getTargetConstant(
12953 Aux & (IsGFX12Plus ? AMDGPU::CPol::ALL : AMDGPU::CPol::ALL_pregfx12),
12954 DL, MVT::i8)); // cpol
12955 Ops.push_back(DAG.getTargetConstant(
12956 Aux & (IsGFX12Plus ? AMDGPU::CPol::SWZ : AMDGPU::CPol::SWZ_pregfx12)
12957 ? 1
12958 : 0,
12959 DL, MVT::i8)); // swz
12960 Ops.push_back(
12961 DAG.getTargetConstant(isAsyncLDSDMA(IntrinsicID), DL, MVT::i8));
12962 Ops.push_back(M0Val.getValue(0)); // Chain
12963 Ops.push_back(M0Val.getValue(1)); // Glue
12964
12965 auto *M = cast<MemSDNode>(Op);
12966 auto *Load = DAG.getMachineNode(Opc, DL, M->getVTList(), Ops);
12967 DAG.setNodeMemRefs(Load, M->memoperands());
12968
12969 return SDValue(Load, 0);
12970 }
12971 // Buffers are handled by LowerBufferFatPointers, and we're going to go
12972 // for "trust me" that the remaining cases are global pointers until
12973 // such time as we can put two mem operands on an intrinsic.
12974 case Intrinsic::amdgcn_load_to_lds:
12975 case Intrinsic::amdgcn_load_async_to_lds:
12976 case Intrinsic::amdgcn_global_load_lds:
12977 case Intrinsic::amdgcn_global_load_async_lds: {
12978 if (!Subtarget->hasVMemToLDSLoad())
12979 return SDValue();
12980
12981 unsigned Opc;
12982 unsigned Size = Op->getConstantOperandVal(4);
12983 switch (Size) {
12984 default:
12985 return SDValue();
12986 case 1:
12987 Opc = AMDGPU::GLOBAL_LOAD_LDS_UBYTE;
12988 break;
12989 case 2:
12990 Opc = AMDGPU::GLOBAL_LOAD_LDS_USHORT;
12991 break;
12992 case 4:
12993 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORD;
12994 break;
12995 case 12:
12996 if (!Subtarget->hasLDSLoadB96_B128())
12997 return SDValue();
12998 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX3;
12999 break;
13000 case 16:
13001 if (!Subtarget->hasLDSLoadB96_B128())
13002 return SDValue();
13003 Opc = AMDGPU::GLOBAL_LOAD_LDS_DWORDX4;
13004 break;
13005 }
13006
13007 SDValue M0Val = copyToM0(DAG, Chain, DL, Op.getOperand(3));
13008
13010
13011 SDValue Addr = Op.getOperand(2); // Global ptr
13012 SDValue VOffset;
13013 // Try to split SAddr and VOffset. Global and LDS pointers share the same
13014 // immediate offset, so we cannot use a regular SelectGlobalSAddr().
13015 if (Addr->isDivergent() && Addr->isAnyAdd()) {
13016 SDValue LHS = Addr.getOperand(0);
13017 SDValue RHS = Addr.getOperand(1);
13018
13019 if (LHS->isDivergent())
13020 std::swap(LHS, RHS);
13021
13022 if (!LHS->isDivergent() && RHS.getOpcode() == ISD::ZERO_EXTEND &&
13023 RHS.getOperand(0).getValueType() == MVT::i32) {
13024 // add (i64 sgpr), (zero_extend (i32 vgpr))
13025 Addr = LHS;
13026 VOffset = RHS.getOperand(0);
13027 }
13028 }
13029
13030 Ops.push_back(Addr);
13031 if (!Addr->isDivergent()) {
13033 if (!VOffset)
13034 VOffset =
13035 SDValue(DAG.getMachineNode(AMDGPU::V_MOV_B32_e32, DL, MVT::i32,
13036 DAG.getTargetConstant(0, DL, MVT::i32)),
13037 0);
13038 Ops.push_back(VOffset);
13039 }
13040
13041 Ops.push_back(Op.getOperand(5)); // Offset
13042
13043 unsigned Aux = Op.getConstantOperandVal(6);
13044 Ops.push_back(DAG.getTargetConstant(Aux & ~AMDGPU::CPol::VIRTUAL_BITS, DL,
13045 MVT::i32)); // CPol
13046 Ops.push_back(
13047 DAG.getTargetConstant(isAsyncLDSDMA(IntrinsicID), DL, MVT::i8));
13048
13049 Ops.push_back(M0Val.getValue(0)); // Chain
13050 Ops.push_back(M0Val.getValue(1)); // Glue
13051
13052 auto *M = cast<MemSDNode>(Op);
13053 auto *Load = DAG.getMachineNode(Opc, DL, Op->getVTList(), Ops);
13054 DAG.setNodeMemRefs(Load, M->memoperands());
13055
13056 return SDValue(Load, 0);
13057 }
13058 case Intrinsic::amdgcn_end_cf:
13059 return SDValue(DAG.getMachineNode(AMDGPU::SI_END_CF, DL, MVT::Other,
13060 Op->getOperand(2), Chain),
13061 0);
13062 case Intrinsic::amdgcn_s_barrier_signal_var: {
13063 // Member count of 0 means to re-use a previous member count,
13064 // which, if the named barrier is statically chosen, means we can use
13065 // the immarg form. Otherwisee, fall through to constructiong M0 as for
13066 // s_barrier_init.
13067 SDValue CntOp = Op->getOperand(3);
13068 auto *CntC = dyn_cast<ConstantSDNode>(CntOp);
13069 if (CntC && CntC->isZero()) {
13070 SDValue Chain = Op->getOperand(0);
13071 SDValue BarOp = Op->getOperand(2);
13073
13074 std::optional<uint64_t> BarVal;
13075 if (auto *C = dyn_cast<ConstantSDNode>(BarOp))
13076 BarVal = C->getZExtValue();
13077 else if (auto *GA = dyn_cast<GlobalAddressSDNode>(BarOp))
13079 *GA->getGlobal(), AMDGPUAS::BARRIER))
13080 BarVal = *Addr + GA->getOffset();
13081
13082 if (BarVal) {
13083 unsigned BarID = *BarVal & 0x3F;
13084 Ops.push_back(DAG.getTargetConstant(BarID, DL, MVT::i32));
13085 Ops.push_back(Chain);
13086 auto *NewMI = DAG.getMachineNode(AMDGPU::S_BARRIER_SIGNAL_IMM, DL,
13087 Op->getVTList(), Ops);
13088 return SDValue(NewMI, 0);
13089 }
13090 }
13091 [[fallthrough]];
13092 }
13093 case Intrinsic::amdgcn_s_barrier_init: {
13094 // these two intrinsics have two operands: barrier pointer and member count
13095 SDValue Chain = Op->getOperand(0);
13097 SDValue BarOp = Op->getOperand(2);
13098 SDValue CntOp = Op->getOperand(3);
13099 SDValue M0Val;
13100 unsigned Opc = IntrinsicID == Intrinsic::amdgcn_s_barrier_init
13101 ? AMDGPU::S_BARRIER_INIT_M0
13102 : AMDGPU::S_BARRIER_SIGNAL_M0;
13103 // extract the BarrierID from bits 0-5 of BarOp
13104 SDValue BarID = DAG.getNode(ISD::AND, DL, MVT::i32, BarOp,
13105 DAG.getConstant(0x3F, DL, MVT::i32));
13106 // Member count should be put into M0[ShAmt:+6]
13107 // Barrier ID should be put into M0[5:0]
13108 SDValue MemberCnt = DAG.getNode(ISD::AND, DL, MVT::i32, CntOp,
13109 DAG.getConstant(0x3F, DL, MVT::i32));
13110 constexpr unsigned ShAmt = 16;
13111 M0Val = DAG.getNode(ISD::SHL, DL, MVT::i32, MemberCnt,
13112 DAG.getShiftAmountConstant(ShAmt, MVT::i32, DL));
13113
13114 M0Val = DAG.getNode(ISD::OR, DL, MVT::i32, M0Val, BarID);
13115
13116 Ops.push_back(copyToM0(DAG, Chain, DL, M0Val).getValue(0));
13117
13118 auto *NewMI = DAG.getMachineNode(Opc, DL, Op->getVTList(), Ops);
13119 return SDValue(NewMI, 0);
13120 }
13121 case Intrinsic::amdgcn_s_wakeup_barrier: {
13122 if (!Subtarget->hasSWakeupBarrier())
13123 return SDValue();
13124 [[fallthrough]];
13125 }
13126 case Intrinsic::amdgcn_s_barrier_join: {
13127 // these three intrinsics have one operand: barrier pointer
13128 SDValue Chain = Op->getOperand(0);
13130 SDValue BarOp = Op->getOperand(2);
13131 unsigned Opc;
13132
13133 if (isa<ConstantSDNode>(BarOp)) {
13134 uint64_t BarVal = cast<ConstantSDNode>(BarOp)->getZExtValue();
13135 switch (IntrinsicID) {
13136 default:
13137 return SDValue();
13138 case Intrinsic::amdgcn_s_barrier_join:
13139 Opc = AMDGPU::S_BARRIER_JOIN_IMM;
13140 break;
13141 case Intrinsic::amdgcn_s_wakeup_barrier:
13142 Opc = AMDGPU::S_WAKEUP_BARRIER_IMM;
13143 break;
13144 }
13145 // extract the BarrierID from bits 0-5 of the immediate
13146 unsigned BarID = BarVal & 0x3F;
13147 SDValue K = DAG.getTargetConstant(BarID, DL, MVT::i32);
13148 Ops.push_back(K);
13149 Ops.push_back(Chain);
13150 } else {
13151 switch (IntrinsicID) {
13152 default:
13153 return SDValue();
13154 case Intrinsic::amdgcn_s_barrier_join:
13155 Opc = AMDGPU::S_BARRIER_JOIN_M0;
13156 break;
13157 case Intrinsic::amdgcn_s_wakeup_barrier:
13158 Opc = AMDGPU::S_WAKEUP_BARRIER_M0;
13159 break;
13160 }
13161 // extract the BarrierID from bits 0-5 of BarOp, copy to M0[5:0]
13162 SDValue M0Val = DAG.getNode(ISD::AND, DL, MVT::i32, BarOp,
13163 DAG.getConstant(0x3F, DL, MVT::i32));
13164 Ops.push_back(copyToM0(DAG, Chain, DL, M0Val).getValue(0));
13165 }
13166
13167 auto *NewMI = DAG.getMachineNode(Opc, DL, Op->getVTList(), Ops);
13168 return SDValue(NewMI, 0);
13169 }
13170 case Intrinsic::amdgcn_s_prefetch_data:
13171 case Intrinsic::amdgcn_s_prefetch_inst: {
13172 // For non-global address space preserve the chain and remove the call.
13174 return Op.getOperand(0);
13175 return Op;
13176 }
13177 case Intrinsic::amdgcn_s_buffer_prefetch_data: {
13178 SDValue Ops[] = {
13179 Chain, bufferRsrcPtrToVector(Op.getOperand(2), DAG),
13180 Op.getOperand(3), // offset
13181 Op.getOperand(4), // length
13182 };
13183
13184 MemSDNode *M = cast<MemSDNode>(Op);
13185 return DAG.getMemIntrinsicNode(AMDGPUISD::SBUFFER_PREFETCH_DATA, DL,
13186 Op->getVTList(), Ops, M->getMemoryVT(),
13187 M->getMemOperand());
13188 }
13189 case Intrinsic::amdgcn_cooperative_atomic_store_32x4B:
13190 case Intrinsic::amdgcn_cooperative_atomic_store_16x8B:
13191 case Intrinsic::amdgcn_cooperative_atomic_store_8x16B: {
13192 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
13193 SDValue Chain = Op->getOperand(0);
13194 SDValue Ptr = Op->getOperand(2);
13195 SDValue Val = Op->getOperand(3);
13196 return DAG.getAtomic(ISD::ATOMIC_STORE, DL, MII->getMemoryVT(), Chain, Val,
13197 Ptr, MII->getMemOperand());
13198 }
13199 case Intrinsic::amdgcn_av_store_b128: {
13200 MemIntrinsicSDNode *MII = cast<MemIntrinsicSDNode>(Op);
13201 SDValue Chain = Op->getOperand(0);
13202 SDValue Ptr = Op->getOperand(2);
13203 SDValue Val = Op->getOperand(3);
13204 return DAG.getStore(Chain, DL, Val, Ptr, MII->getMemOperand());
13205 }
13206 default: {
13207 if (const AMDGPU::ImageDimIntrinsicInfo *ImageDimIntr =
13209 return lowerImage(Op, ImageDimIntr, DAG, true);
13210
13211 return Op;
13212 }
13213 }
13214}
13215
13216// Return whether the operation has NoUnsignedWrap property.
13217static bool isNoUnsignedWrap(SDValue Addr) {
13218 return (Addr.getOpcode() == ISD::ADD &&
13219 Addr->getFlags().hasNoUnsignedWrap()) ||
13220 Addr->getOpcode() == ISD::OR;
13221}
13222
13224 EVT PtrVT) const {
13225 return PtrVT == MVT::i64;
13226}
13227
13229 EVT PtrVT) const {
13230 return true;
13231}
13232
13233// The raw.(t)buffer and struct.(t)buffer intrinsics have two offset args:
13234// offset (the offset that is included in bounds checking and swizzling, to be
13235// split between the instruction's voffset and immoffset fields) and soffset
13236// (the offset that is excluded from bounds checking and swizzling, to go in
13237// the instruction's soffset field). This function takes the first kind of
13238// offset and figures out how to split it between voffset and immoffset.
13239std::pair<SDValue, SDValue>
13240SITargetLowering::splitBufferOffsets(SDValue Offset, SelectionDAG &DAG) const {
13241 SDLoc DL(Offset);
13242 const unsigned MaxImm = SIInstrInfo::getMaxMUBUFImmOffset(*Subtarget);
13243 SDValue N0 = Offset;
13244 ConstantSDNode *C1 = nullptr;
13245
13246 if ((C1 = dyn_cast<ConstantSDNode>(N0)))
13247 N0 = SDValue();
13248 else if (DAG.isBaseWithConstantOffset(N0)) {
13249 // On GFX1250+, voffset and immoffset are zero-extended from 32 bits before
13250 // being added, so we can only safely match a 32-bit addition with no
13251 // unsigned overflow.
13252 bool CheckNUW = Subtarget->hasGFX1250Insts();
13253 if (!CheckNUW || isNoUnsignedWrap(N0)) {
13254 C1 = cast<ConstantSDNode>(N0.getOperand(1));
13255 N0 = N0.getOperand(0);
13256 }
13257 }
13258
13259 if (C1) {
13260 unsigned ImmOffset = C1->getZExtValue();
13261 // If the immediate value is too big for the immoffset field, put only bits
13262 // that would normally fit in the immoffset field. The remaining value that
13263 // is copied/added for the voffset field is a large power of 2, and it
13264 // stands more chance of being CSEd with the copy/add for another similar
13265 // load/store.
13266 // However, do not do that rounding down if that is a negative
13267 // number, as it appears to be illegal to have a negative offset in the
13268 // vgpr, even if adding the immediate offset makes it positive.
13269 unsigned Overflow = ImmOffset & ~MaxImm;
13270 ImmOffset -= Overflow;
13271 if ((int32_t)Overflow < 0) {
13272 Overflow += ImmOffset;
13273 ImmOffset = 0;
13274 }
13275 C1 = cast<ConstantSDNode>(DAG.getTargetConstant(ImmOffset, DL, MVT::i32));
13276 if (Overflow) {
13277 auto OverflowVal = DAG.getConstant(Overflow, DL, MVT::i32);
13278 if (!N0)
13279 N0 = OverflowVal;
13280 else {
13281 SDValue Ops[] = {N0, OverflowVal};
13282 N0 = DAG.getNode(ISD::ADD, DL, MVT::i32, Ops);
13283 }
13284 }
13285 }
13286 if (!N0)
13287 N0 = DAG.getConstant(0, DL, MVT::i32);
13288 if (!C1)
13289 C1 = cast<ConstantSDNode>(DAG.getTargetConstant(0, DL, MVT::i32));
13290 return {N0, SDValue(C1, 0)};
13291}
13292
13293// Analyze a combined offset from an amdgcn_s_buffer_load intrinsic and store
13294// the three offsets (voffset, soffset and instoffset) into the SDValue[3] array
13295// pointed to by Offsets.
13296void SITargetLowering::setBufferOffsets(SDValue CombinedOffset,
13297 SelectionDAG &DAG, SDValue *Offsets,
13298 Align Alignment) const {
13299 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
13300 SDLoc DL(CombinedOffset);
13301 if (auto *C = dyn_cast<ConstantSDNode>(CombinedOffset)) {
13302 uint32_t Imm = C->getZExtValue();
13303 uint32_t SOffset, ImmOffset;
13304 if (TII->splitMUBUFOffset(Imm, SOffset, ImmOffset, Alignment)) {
13305 Offsets[0] = DAG.getConstant(0, DL, MVT::i32);
13306 Offsets[1] = DAG.getConstant(SOffset, DL, MVT::i32);
13307 Offsets[2] = DAG.getTargetConstant(ImmOffset, DL, MVT::i32);
13308 return;
13309 }
13310 }
13311 if (DAG.isBaseWithConstantOffset(CombinedOffset)) {
13312 // On GFX1250+, voffset and immoffset are zero-extended from 32 bits before
13313 // being added, so we can only safely match a 32-bit addition with no
13314 // unsigned overflow.
13315 bool CheckNUW = Subtarget->hasGFX1250Insts();
13316 SDValue N0 = CombinedOffset.getOperand(0);
13317 SDValue N1 = CombinedOffset.getOperand(1);
13318 uint32_t SOffset, ImmOffset;
13319 int Offset = cast<ConstantSDNode>(N1)->getSExtValue();
13320 if (Offset >= 0 && (!CheckNUW || isNoUnsignedWrap(CombinedOffset)) &&
13321 TII->splitMUBUFOffset(Offset, SOffset, ImmOffset, Alignment)) {
13322 Offsets[0] = N0;
13323 Offsets[1] = DAG.getConstant(SOffset, DL, MVT::i32);
13324 Offsets[2] = DAG.getTargetConstant(ImmOffset, DL, MVT::i32);
13325 return;
13326 }
13327 }
13328
13329 SDValue SOffsetZero = Subtarget->hasRestrictedSOffset()
13330 ? DAG.getRegister(AMDGPU::SGPR_NULL, MVT::i32)
13331 : DAG.getConstant(0, DL, MVT::i32);
13332
13333 Offsets[0] = CombinedOffset;
13334 Offsets[1] = SOffsetZero;
13335 Offsets[2] = DAG.getTargetConstant(0, DL, MVT::i32);
13336}
13337
13338SDValue SITargetLowering::bufferRsrcPtrToVector(SDValue MaybePointer,
13339 SelectionDAG &DAG) const {
13340 if (!MaybePointer.getValueType().isScalarInteger())
13341 return MaybePointer;
13342
13343 SDValue Rsrc = DAG.getBitcast(MVT::v4i32, MaybePointer);
13344 return Rsrc;
13345}
13346
13347// Wrap a global or flat pointer into a buffer intrinsic using the flags
13348// specified in the intrinsic.
13349SDValue SITargetLowering::lowerPointerAsRsrcIntrin(SDNode *Op,
13350 SelectionDAG &DAG) const {
13351 SDLoc Loc(Op);
13352
13353 SDValue Pointer = Op->getOperand(1);
13354 SDValue Stride = Op->getOperand(2);
13355 SDValue NumRecords = Op->getOperand(3);
13356 SDValue Flags = Op->getOperand(4);
13357
13358 SDValue ExtStride = DAG.getAnyExtOrTrunc(Stride, Loc, MVT::i32);
13359 SDValue Rsrc;
13360
13361 if (Subtarget->getBufferResourceNumRecordsWidth() == 45) {
13362 NumRecords = DAG.getZExtOrTrunc(NumRecords, Loc, MVT::i64);
13363 NumRecords = DAG.getNode(ISD::AND, Loc, MVT::i64, NumRecords,
13364 DAG.getConstant((1ULL << 45) - 1, Loc, MVT::i64));
13365 SDValue Zero = DAG.getConstant(0, Loc, MVT::i32);
13366 // Build the lower 64-bit value, which has a 57-bit base and the lower 7-bit
13367 // num_records.
13368 SDValue ExtPointer = DAG.getAnyExtOrTrunc(Pointer, Loc, MVT::i64);
13369 SDValue NumRecordsLHS =
13370 DAG.getNode(ISD::SHL, Loc, MVT::i64, NumRecords,
13371 DAG.getShiftAmountConstant(57, MVT::i32, Loc));
13372 SDValue LowHalf =
13373 DAG.getNode(ISD::OR, Loc, MVT::i64, ExtPointer, NumRecordsLHS);
13374
13375 // Build the higher 64-bit value, which has the higher 38-bit num_records,
13376 // 6-bit zero (omit), 16-bit stride and scale and 4-bit flag.
13377 SDValue NumRecordsRHS =
13378 DAG.getNode(ISD::SRL, Loc, MVT::i64, NumRecords,
13379 DAG.getShiftAmountConstant(7, MVT::i32, Loc));
13380 SDValue ShiftedStride =
13381 DAG.getNode(ISD::SHL, Loc, MVT::i32, ExtStride,
13382 DAG.getShiftAmountConstant(12, MVT::i32, Loc));
13383 SDValue ExtShiftedStrideVec =
13384 DAG.getNode(ISD::BUILD_VECTOR, Loc, MVT::v2i32, Zero, ShiftedStride);
13385 SDValue ExtShiftedStride =
13386 DAG.getNode(ISD::BITCAST, Loc, MVT::i64, ExtShiftedStrideVec);
13387 SDValue ShiftedFlags =
13388 DAG.getNode(ISD::SHL, Loc, MVT::i32, Flags,
13389 DAG.getShiftAmountConstant(28, MVT::i32, Loc));
13390 SDValue ExtShiftedFlagsVec =
13391 DAG.getNode(ISD::BUILD_VECTOR, Loc, MVT::v2i32, Zero, ShiftedFlags);
13392 SDValue ExtShiftedFlags =
13393 DAG.getNode(ISD::BITCAST, Loc, MVT::i64, ExtShiftedFlagsVec);
13394 SDValue CombinedFields =
13395 DAG.getNode(ISD::OR, Loc, MVT::i64, NumRecordsRHS, ExtShiftedStride);
13396 SDValue HighHalf =
13397 DAG.getNode(ISD::OR, Loc, MVT::i64, CombinedFields, ExtShiftedFlags);
13398
13399 Rsrc = DAG.getNode(ISD::BUILD_VECTOR, Loc, MVT::v2i64, LowHalf, HighHalf);
13400 } else {
13401 NumRecords = DAG.getZExtOrTrunc(NumRecords, Loc, MVT::i32);
13402 auto [LowHalf, HighHalf] =
13403 DAG.SplitScalar(Pointer, Loc, MVT::i32, MVT::i32);
13404 SDValue Mask = DAG.getConstant(0x0000ffff, Loc, MVT::i32);
13405 SDValue Masked = DAG.getNode(ISD::AND, Loc, MVT::i32, HighHalf, Mask);
13406 SDValue ShiftedStride =
13407 DAG.getNode(ISD::SHL, Loc, MVT::i32, ExtStride,
13408 DAG.getShiftAmountConstant(16, MVT::i32, Loc));
13409 SDValue NewHighHalf =
13410 DAG.getNode(ISD::OR, Loc, MVT::i32, Masked, ShiftedStride);
13411
13412 Rsrc = DAG.getNode(ISD::BUILD_VECTOR, Loc, MVT::v4i32, LowHalf, NewHighHalf,
13413 NumRecords, Flags);
13414 }
13415
13416 SDValue RsrcPtr = DAG.getNode(ISD::BITCAST, Loc, MVT::i128, Rsrc);
13417 return RsrcPtr;
13418}
13419
13420// Handle 8 bit and 16 bit buffer loads
13421SDValue SITargetLowering::handleByteShortBufferLoads(SelectionDAG &DAG,
13422 EVT LoadVT, SDLoc DL,
13424 MachineMemOperand *MMO,
13425 bool IsTFE) const {
13426 EVT IntVT = LoadVT.changeTypeToInteger();
13427
13428 if (IsTFE) {
13429 unsigned Opc = (LoadVT.getScalarType() == MVT::i8)
13430 ? AMDGPUISD::BUFFER_LOAD_UBYTE_TFE
13431 : AMDGPUISD::BUFFER_LOAD_USHORT_TFE;
13433 MachineMemOperand *OpMMO = MF.getMachineMemOperand(MMO, 0, 8);
13434 SDVTList VTs = DAG.getVTList(MVT::v2i32, MVT::Other);
13435 SDValue Op = getMemIntrinsicNode(Opc, DL, VTs, Ops, MVT::v2i32, OpMMO, DAG);
13436 SDValue Status = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, Op,
13437 DAG.getConstant(1, DL, MVT::i32));
13438 SDValue Data = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, Op,
13439 DAG.getConstant(0, DL, MVT::i32));
13440 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, IntVT, Data);
13441 SDValue Value = DAG.getNode(ISD::BITCAST, DL, LoadVT, Trunc);
13442 return DAG.getMergeValues({Value, Status, SDValue(Op.getNode(), 1)}, DL);
13443 }
13444
13445 unsigned Opc = LoadVT.getScalarType() == MVT::i8
13446 ? AMDGPUISD::BUFFER_LOAD_UBYTE
13447 : AMDGPUISD::BUFFER_LOAD_USHORT;
13448
13449 SDVTList ResList = DAG.getVTList(MVT::i32, MVT::Other);
13450 SDValue BufferLoad =
13451 DAG.getMemIntrinsicNode(Opc, DL, ResList, Ops, IntVT, MMO);
13452 SDValue LoadVal = DAG.getNode(ISD::TRUNCATE, DL, IntVT, BufferLoad);
13453 LoadVal = DAG.getNode(ISD::BITCAST, DL, LoadVT, LoadVal);
13454
13455 return DAG.getMergeValues({LoadVal, BufferLoad.getValue(1)}, DL);
13456}
13457
13458// Handle 8 bit and 16 bit buffer stores
13459SDValue SITargetLowering::handleByteShortBufferStores(SelectionDAG &DAG,
13460 EVT VDataType, SDLoc DL,
13461 SDValue Ops[],
13462 MemSDNode *M) const {
13463 if (VDataType == MVT::f16 || VDataType == MVT::bf16)
13464 Ops[1] = DAG.getNode(ISD::BITCAST, DL, MVT::i16, Ops[1]);
13465
13466 SDValue BufferStoreExt = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, Ops[1]);
13467 Ops[1] = BufferStoreExt;
13468 unsigned Opc = (VDataType == MVT::i8) ? AMDGPUISD::BUFFER_STORE_BYTE
13469 : AMDGPUISD::BUFFER_STORE_SHORT;
13470 ArrayRef<SDValue> OpsRef = ArrayRef(&Ops[0], 9);
13471 return DAG.getMemIntrinsicNode(Opc, DL, M->getVTList(), OpsRef, VDataType,
13472 M->getMemOperand());
13473}
13474
13476 SDValue Op, const SDLoc &SL, EVT VT) {
13477 if (VT.bitsLT(Op.getValueType()))
13478 return DAG.getNode(ISD::TRUNCATE, SL, VT, Op);
13479
13480 switch (ExtType) {
13481 case ISD::SEXTLOAD:
13482 return DAG.getNode(ISD::SIGN_EXTEND, SL, VT, Op);
13483 case ISD::ZEXTLOAD:
13484 return DAG.getNode(ISD::ZERO_EXTEND, SL, VT, Op);
13485 case ISD::EXTLOAD:
13486 return DAG.getNode(ISD::ANY_EXTEND, SL, VT, Op);
13487 case ISD::NON_EXTLOAD:
13488 return Op;
13489 }
13490
13491 llvm_unreachable("invalid ext type");
13492}
13493
13494// Try to turn 8 and 16-bit scalar loads into SMEM eligible 32-bit loads.
13495// TODO: Skip this on GFX12 which does have scalar sub-dword loads.
13496SDValue SITargetLowering::widenLoad(LoadSDNode *Ld,
13497 DAGCombinerInfo &DCI) const {
13498 SelectionDAG &DAG = DCI.DAG;
13499 if (Ld->getAlign() < Align(4) || Ld->isDivergent())
13500 return SDValue();
13501
13502 // FIXME: Constant loads should all be marked invariant.
13503 unsigned AS = Ld->getAddressSpace();
13504 if (AS != AMDGPUAS::CONSTANT_ADDRESS &&
13506 (AS != AMDGPUAS::GLOBAL_ADDRESS || !Ld->isInvariant()))
13507 return SDValue();
13508
13509 // Don't do this early, since it may interfere with adjacent load merging for
13510 // illegal types. We can avoid losing alignment information for exotic types
13511 // pre-legalize.
13512 EVT MemVT = Ld->getMemoryVT();
13513 if ((MemVT.isSimple() && !DCI.isAfterLegalizeDAG()) ||
13514 MemVT.getSizeInBits() >= 32)
13515 return SDValue();
13516
13517 SDLoc SL(Ld);
13518
13519 assert((!MemVT.isVector() || Ld->getExtensionType() == ISD::NON_EXTLOAD) &&
13520 "unexpected vector extload");
13521
13522 // TODO: Drop only high part of range.
13523 SDValue Ptr = Ld->getBasePtr();
13524 SDValue NewLoad = DAG.getLoad(
13525 ISD::UNINDEXED, ISD::NON_EXTLOAD, MVT::i32, SL, Ld->getChain(), Ptr,
13526 Ld->getOffset(), Ld->getPointerInfo(), MVT::i32, Ld->getAlign(),
13527 Ld->getMemOperand()->getFlags(), Ld->getAAInfo()); // Drop ranges
13528
13529 EVT TruncVT = EVT::getIntegerVT(*DAG.getContext(), MemVT.getSizeInBits());
13530 if (MemVT.isFloatingPoint()) {
13531 assert(Ld->getExtensionType() == ISD::NON_EXTLOAD &&
13532 "unexpected fp extload");
13533 TruncVT = MemVT.changeTypeToInteger();
13534 }
13535
13536 SDValue Cvt = NewLoad;
13537 if (Ld->getExtensionType() == ISD::SEXTLOAD) {
13538 Cvt = DAG.getNode(ISD::SIGN_EXTEND_INREG, SL, MVT::i32, NewLoad,
13539 DAG.getValueType(TruncVT));
13540 } else if (Ld->getExtensionType() == ISD::ZEXTLOAD ||
13541 Ld->getExtensionType() == ISD::NON_EXTLOAD) {
13542 Cvt = DAG.getZeroExtendInReg(NewLoad, SL, TruncVT);
13543 } else {
13544 assert(Ld->getExtensionType() == ISD::EXTLOAD);
13545 }
13546
13547 EVT VT = Ld->getValueType(0);
13548 EVT IntVT = EVT::getIntegerVT(*DAG.getContext(), VT.getSizeInBits());
13549
13550 DCI.AddToWorklist(Cvt.getNode());
13551
13552 // We may need to handle exotic cases, such as i16->i64 extloads, so insert
13553 // the appropriate extension from the 32-bit load.
13554 Cvt = getLoadExtOrTrunc(DAG, Ld->getExtensionType(), Cvt, SL, IntVT);
13555 DCI.AddToWorklist(Cvt.getNode());
13556
13557 // Handle conversion back to floating point if necessary.
13558 Cvt = DAG.getNode(ISD::BITCAST, SL, VT, Cvt);
13559
13560 return DAG.getMergeValues({Cvt, NewLoad.getValue(1)}, SL);
13561}
13562
13564 const SIMachineFunctionInfo &Info) {
13565 // TODO: Should check if the address can definitely not access stack.
13566 if (Info.isEntryFunction())
13567 return Info.getUserSGPRInfo().hasFlatScratchInit();
13568 return true;
13569}
13570
13571SDValue SITargetLowering::LowerLOAD(SDValue Op, SelectionDAG &DAG) const {
13572 SDLoc DL(Op);
13573 LoadSDNode *Load = cast<LoadSDNode>(Op);
13574 ISD::LoadExtType ExtType = Load->getExtensionType();
13575 EVT MemVT = Load->getMemoryVT();
13576 MachineMemOperand *MMO = Load->getMemOperand();
13577
13578 if (ExtType == ISD::NON_EXTLOAD && MemVT.getSizeInBits() < 32) {
13579 // Legalize uniform 16-bit loads to i16 = trunc (zextload i16->i32)
13580 // to match subword load patterns.
13581 // Only do this for loads that can use scalar subword load instructions.
13582 if (!MemVT.isVector() && MemVT.getSizeInBits() == 16 &&
13583 isTypeLegal(MemVT) && Subtarget->hasScalarSubwordLoads() &&
13585 SDValue Chain = Load->getChain();
13586 SDValue BasePtr = Load->getBasePtr();
13587
13588 // Load as i16 and zero-extend to i32 (matches S_LOAD_U16 behavior)
13589 SDValue NewLD = DAG.getExtLoad(ISD::ZEXTLOAD, DL, MVT::i32, Chain,
13590 BasePtr, MVT::i16, MMO);
13591
13592 // Truncate back to i16
13593 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, DL, MVT::i16, NewLD);
13594
13595 // For f16/bf16, bitcast from i16 to the original fp type
13596 SDValue Result = (MemVT == MVT::i16)
13597 ? Trunc
13598 : DAG.getNode(ISD::BITCAST, DL, MemVT, Trunc);
13599
13600 SDValue Ops[] = {Result, NewLD.getValue(1)};
13601 return DAG.getMergeValues(Ops, DL);
13602 }
13603
13604 if (MemVT == MVT::i16 && isTypeLegal(MVT::i16))
13605 return SDValue();
13606
13607 // FIXME: Copied from PPC
13608 // First, load into 32 bits, then truncate to 1 bit.
13609
13610 SDValue Chain = Load->getChain();
13611 SDValue BasePtr = Load->getBasePtr();
13612
13613 EVT RealMemVT = (MemVT == MVT::i1) ? MVT::i8 : MVT::i16;
13614
13615 SDValue NewLD = DAG.getExtLoad(ISD::EXTLOAD, DL, MVT::i32, Chain, BasePtr,
13616 RealMemVT, MMO);
13617
13618 if (!MemVT.isVector()) {
13619 SDValue Ops[] = {DAG.getNode(ISD::TRUNCATE, DL, MemVT, NewLD),
13620 NewLD.getValue(1)};
13621
13622 return DAG.getMergeValues(Ops, DL);
13623 }
13624
13626 for (unsigned I = 0, N = MemVT.getVectorNumElements(); I != N; ++I) {
13627 SDValue Elt = DAG.getNode(ISD::SRL, DL, MVT::i32, NewLD,
13628 DAG.getConstant(I, DL, MVT::i32));
13629
13630 Elts.push_back(DAG.getNode(ISD::TRUNCATE, DL, MVT::i1, Elt));
13631 }
13632
13633 SDValue Ops[] = {DAG.getBuildVector(MemVT, DL, Elts), NewLD.getValue(1)};
13634
13635 return DAG.getMergeValues(Ops, DL);
13636 }
13637
13638 if (!MemVT.isVector())
13639 return SDValue();
13640
13641 assert(Op.getValueType().getVectorElementType() == MVT::i32 &&
13642 "Custom lowering for non-i32 vectors hasn't been implemented.");
13643
13644 Align Alignment = Load->getAlign();
13645 unsigned AS = Load->getAddressSpace();
13646 if (Subtarget->hasLDSMisalignedBugInWGPMode() &&
13647 AS == AMDGPUAS::FLAT_ADDRESS &&
13648 Alignment.value() < MemVT.getStoreSize() && MemVT.getSizeInBits() > 32) {
13649 return SplitVectorLoad(Op, DAG);
13650 }
13651
13653 SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
13654 // If there is a possibility that flat instruction access scratch memory
13655 // then we need to use the same legalization rules we use for private.
13656 if (AS == AMDGPUAS::FLAT_ADDRESS &&
13657 !Subtarget->hasMultiDwordFlatScratchAddressing())
13658 AS = addressMayBeAccessedAsPrivate(Load->getMemOperand(), *MFI)
13661
13662 unsigned NumElements = MemVT.getVectorNumElements();
13663
13664 if (AS == AMDGPUAS::CONSTANT_ADDRESS ||
13666 (AS == AMDGPUAS::GLOBAL_ADDRESS && Load->isSimple() &&
13667 (Load->isInvariant() || isMemOpHasNoClobberedMemOperand(Load)))) {
13668 if ((!Op->isDivergent() || AMDGPU::isUniformMMO(MMO)) &&
13669 Alignment >= Align(4) && NumElements < 32) {
13670 if (MemVT.isPow2VectorType() ||
13671 (Subtarget->hasScalarDwordx3Loads() && NumElements == 3))
13672 return SDValue();
13673 return WidenOrSplitVectorLoad(Op, DAG);
13674 }
13675 // Non-uniform loads will be selected to MUBUF instructions, so they
13676 // have the same legalization requirements as global and private
13677 // loads.
13678 //
13679 }
13680 if (AS == AMDGPUAS::CONSTANT_ADDRESS ||
13683 if (NumElements > 4)
13684 return SplitVectorLoad(Op, DAG);
13685 // v3 loads not supported on SI.
13686 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
13687 return WidenOrSplitVectorLoad(Op, DAG);
13688
13689 // v3 and v4 loads are supported for private and global memory.
13690 return SDValue();
13691 }
13692 if (AS == AMDGPUAS::PRIVATE_ADDRESS) {
13693 // Depending on the setting of the private_element_size field in the
13694 // resource descriptor, we can only make private accesses up to a certain
13695 // size.
13696 switch (Subtarget->getMaxPrivateElementSize()) {
13697 case 4: {
13698 auto [Op0, Op1] = scalarizeVectorLoad(Load, DAG);
13699 return DAG.getMergeValues({Op0, Op1}, DL);
13700 }
13701 case 8:
13702 if (NumElements > 2)
13703 return SplitVectorLoad(Op, DAG);
13704 return SDValue();
13705 case 16:
13706 // Same as global/flat
13707 if (NumElements > 4)
13708 return SplitVectorLoad(Op, DAG);
13709 // v3 loads not supported on SI.
13710 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
13711 return WidenOrSplitVectorLoad(Op, DAG);
13712
13713 return SDValue();
13714 default:
13715 llvm_unreachable("unsupported private_element_size");
13716 }
13717 } else if (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::REGION_ADDRESS) {
13718 unsigned Fast = 0;
13719 auto Flags = Load->getMemOperand()->getFlags();
13721 Load->getAlign(), Flags, &Fast) &&
13722 Fast > 1)
13723 return SDValue();
13724
13725 if (MemVT.isVector())
13726 return SplitVectorLoad(Op, DAG);
13727 }
13728
13730 MemVT, *Load->getMemOperand())) {
13731 auto [Op0, Op1] = expandUnalignedLoad(Load, DAG);
13732 return DAG.getMergeValues({Op0, Op1}, DL);
13733 }
13734
13735 return SDValue();
13736}
13737
13738SDValue SITargetLowering::LowerSELECT(SDValue Op, SelectionDAG &DAG) const {
13739 EVT VT = Op.getValueType();
13740 if (VT.getSizeInBits() == 128 || VT.getSizeInBits() == 256 ||
13741 VT.getSizeInBits() == 512)
13742 return splitTernaryVectorOp(Op, DAG);
13743
13744 assert(VT.getSizeInBits() == 64);
13745
13746 SDLoc DL(Op);
13747 SDValue Cond = DAG.getFreeze(Op.getOperand(0));
13748
13749 SDValue Zero = DAG.getConstant(0, DL, MVT::i32);
13750 SDValue One = DAG.getConstant(1, DL, MVT::i32);
13751
13752 SDValue LHS = DAG.getNode(ISD::BITCAST, DL, MVT::v2i32, Op.getOperand(1));
13753 SDValue RHS = DAG.getNode(ISD::BITCAST, DL, MVT::v2i32, Op.getOperand(2));
13754
13755 SDValue Lo0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, LHS, Zero);
13756 SDValue Lo1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, RHS, Zero);
13757
13758 SDValue Lo = DAG.getSelect(DL, MVT::i32, Cond, Lo0, Lo1);
13759
13760 SDValue Hi0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, LHS, One);
13761 SDValue Hi1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::i32, RHS, One);
13762
13763 SDValue Hi = DAG.getSelect(DL, MVT::i32, Cond, Hi0, Hi1);
13764
13765 SDValue Res = DAG.getBuildVector(MVT::v2i32, DL, {Lo, Hi});
13766 return DAG.getNode(ISD::BITCAST, DL, VT, Res);
13767}
13768
13769// Catch division cases where we can use shortcuts with rcp and rsq
13770// instructions.
13771SDValue SITargetLowering::lowerFastUnsafeFDIV(SDValue Op,
13772 SelectionDAG &DAG) const {
13773 SDLoc SL(Op);
13774 SDValue LHS = Op.getOperand(0);
13775 SDValue RHS = Op.getOperand(1);
13776 EVT VT = Op.getValueType();
13777 const SDNodeFlags Flags = Op->getFlags();
13778
13779 bool AllowInaccurateRcp = Flags.hasApproximateFuncs();
13780
13781 if (const ConstantFPSDNode *CLHS = dyn_cast<ConstantFPSDNode>(LHS)) {
13782 // Without !fpmath accuracy information, we can't do more because we don't
13783 // know exactly whether rcp is accurate enough to meet !fpmath requirement.
13784 // f16 is always accurate enough
13785 if (!AllowInaccurateRcp && VT != MVT::f16 && VT != MVT::bf16)
13786 return SDValue();
13787
13788 if (CLHS->isOne()) {
13789 // v_rcp_f32 and v_rsq_f32 do not support denormals, and according to
13790 // the CI documentation has a worst case error of 1 ulp.
13791 // OpenCL requires <= 2.5 ulp for 1.0 / x, so it should always be OK to
13792 // use it as long as we aren't trying to use denormals.
13793 //
13794 // v_rcp_f16 and v_rsq_f16 DO support denormals and 0.51ulp.
13795
13796 // 1.0 / sqrt(x) -> rsq(x)
13797
13798 // XXX - Is afn sufficient to do this for f64? The maximum ULP
13799 // error seems really high at 2^29 ULP.
13800 // 1.0 / x -> rcp(x)
13801 return DAG.getNode(AMDGPUISD::RCP, SL, VT, RHS);
13802 }
13803
13804 // Same as for 1.0, but expand the sign out of the constant.
13805 if (CLHS->isMinusOne()) {
13806 // -1.0 / x -> rcp (fneg x)
13807 SDValue FNegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
13808 return DAG.getNode(AMDGPUISD::RCP, SL, VT, FNegRHS);
13809 }
13810 }
13811
13812 // For f16 and bf16 require afn or arcp.
13813 // For f32 require afn.
13814 if (!AllowInaccurateRcp &&
13815 ((VT != MVT::f16 && VT != MVT::bf16) || !Flags.hasAllowReciprocal()))
13816 return SDValue();
13817
13818 // Turn into multiply by the reciprocal.
13819 // x / y -> x * (1.0 / y)
13820 SDValue Recip = DAG.getNode(AMDGPUISD::RCP, SL, VT, RHS);
13821 return DAG.getNode(ISD::FMUL, SL, VT, LHS, Recip, Flags);
13822}
13823
13824SDValue SITargetLowering::lowerFastUnsafeFDIV64(SDValue Op,
13825 SelectionDAG &DAG) const {
13826 SDLoc SL(Op);
13827 SDValue X = Op.getOperand(0);
13828 SDValue Y = Op.getOperand(1);
13829 EVT VT = Op.getValueType();
13830 const SDNodeFlags Flags = Op->getFlags();
13831
13832 bool AllowInaccurateDiv = Flags.hasApproximateFuncs();
13833 if (!AllowInaccurateDiv)
13834 return SDValue();
13835
13836 const ConstantFPSDNode *CLHS = dyn_cast<ConstantFPSDNode>(X);
13837 bool IsNegRcp = CLHS && CLHS->isMinusOne();
13838
13839 // Pull out the negation so it folds for free into the source modifiers.
13840 if (IsNegRcp)
13841 X = DAG.getConstantFP(1.0, SL, VT);
13842
13843 SDValue NegY = IsNegRcp ? Y : DAG.getNode(ISD::FNEG, SL, VT, Y);
13844 SDValue One = DAG.getConstantFP(1.0, SL, VT);
13845
13846 SDValue R = DAG.getNode(AMDGPUISD::RCP, SL, VT, Y);
13847 if (IsNegRcp)
13848 R = DAG.getNode(ISD::FNEG, SL, VT, R);
13849
13850 SDValue Tmp0 = DAG.getNode(ISD::FMA, SL, VT, NegY, R, One);
13851
13852 R = DAG.getNode(ISD::FMA, SL, VT, Tmp0, R, R);
13853 SDValue Tmp1 = DAG.getNode(ISD::FMA, SL, VT, NegY, R, One);
13854 R = DAG.getNode(ISD::FMA, SL, VT, Tmp1, R, R);
13855
13856 // Skip the last 2 correction terms for reciprocal.
13857 if (IsNegRcp || (CLHS && CLHS->isOne()))
13858 return R;
13859
13860 SDValue Ret = DAG.getNode(ISD::FMUL, SL, VT, X, R);
13861 SDValue Tmp2 = DAG.getNode(ISD::FMA, SL, VT, NegY, Ret, X);
13862 return DAG.getNode(ISD::FMA, SL, VT, Tmp2, R, Ret);
13863}
13864
13865static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL,
13866 EVT VT, SDValue A, SDValue B, SDValue GlueChain,
13867 SDNodeFlags Flags) {
13868 if (GlueChain->getNumValues() <= 1) {
13869 return DAG.getNode(Opcode, SL, VT, A, B, Flags);
13870 }
13871
13872 assert(GlueChain->getNumValues() == 3);
13873
13874 SDVTList VTList = DAG.getVTList(VT, MVT::Other, MVT::Glue);
13875 switch (Opcode) {
13876 default:
13877 llvm_unreachable("no chain equivalent for opcode");
13878 case ISD::FMUL:
13879 Opcode = AMDGPUISD::FMUL_W_CHAIN;
13880 break;
13881 }
13882
13883 return DAG.getNode(Opcode, SL, VTList,
13884 {GlueChain.getValue(1), A, B, GlueChain.getValue(2)},
13885 Flags);
13886}
13887
13888static SDValue getFPTernOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL,
13889 EVT VT, SDValue A, SDValue B, SDValue C,
13890 SDValue GlueChain, SDNodeFlags Flags) {
13891 if (GlueChain->getNumValues() <= 1) {
13892 return DAG.getNode(Opcode, SL, VT, {A, B, C}, Flags);
13893 }
13894
13895 assert(GlueChain->getNumValues() == 3);
13896
13897 SDVTList VTList = DAG.getVTList(VT, MVT::Other, MVT::Glue);
13898 switch (Opcode) {
13899 default:
13900 llvm_unreachable("no chain equivalent for opcode");
13901 case ISD::FMA:
13902 Opcode = AMDGPUISD::FMA_W_CHAIN;
13903 break;
13904 }
13905
13906 return DAG.getNode(Opcode, SL, VTList,
13907 {GlueChain.getValue(1), A, B, C, GlueChain.getValue(2)},
13908 Flags);
13909}
13910
13911SDValue SITargetLowering::LowerFDIV16(SDValue Op, SelectionDAG &DAG) const {
13912 if (SDValue FastLowered = lowerFastUnsafeFDIV(Op, DAG))
13913 return FastLowered;
13914
13915 SDLoc SL(Op);
13916 EVT VT = Op.getValueType();
13917 SDValue LHS = Op.getOperand(0);
13918 SDValue RHS = Op.getOperand(1);
13919
13920 SDValue LHSExt = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, LHS);
13921 SDValue RHSExt = DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, RHS);
13922
13923 if (VT == MVT::bf16) {
13924 SDValue ExtDiv =
13925 DAG.getNode(ISD::FDIV, SL, MVT::f32, LHSExt, RHSExt, Op->getFlags());
13926 return DAG.getNode(ISD::FP_ROUND, SL, MVT::bf16, ExtDiv,
13927 DAG.getTargetConstant(0, SL, MVT::i32));
13928 }
13929
13930 assert(VT == MVT::f16);
13931
13932 // a32.u = opx(V_CVT_F32_F16, a.u); // CVT to F32
13933 // b32.u = opx(V_CVT_F32_F16, b.u); // CVT to F32
13934 // r32.u = opx(V_RCP_F32, b32.u); // rcp = 1 / d
13935 // q32.u = opx(V_MUL_F32, a32.u, r32.u); // q = n * rcp
13936 // e32.u = opx(V_MAD_F32, (b32.u^_neg32), q32.u, a32.u); // err = -d * q + n
13937 // q32.u = opx(V_MAD_F32, e32.u, r32.u, q32.u); // q = n * rcp
13938 // e32.u = opx(V_MAD_F32, (b32.u^_neg32), q32.u, a32.u); // err = -d * q + n
13939 // tmp.u = opx(V_MUL_F32, e32.u, r32.u);
13940 // tmp.u = opx(V_AND_B32, tmp.u, 0xff800000)
13941 // q32.u = opx(V_ADD_F32, tmp.u, q32.u);
13942 // q16.u = opx(V_CVT_F16_F32, q32.u);
13943 // q16.u = opx(V_DIV_FIXUP_F16, q16.u, b.u, a.u); // q = touchup(q, d, n)
13944
13945 // We will use ISD::FMA on targets that don't support ISD::FMAD.
13946 unsigned FMADOpCode =
13948 SDValue NegRHSExt = DAG.getNode(ISD::FNEG, SL, MVT::f32, RHSExt);
13949 SDValue Rcp =
13950 DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32, RHSExt, Op->getFlags());
13951 SDValue Quot =
13952 DAG.getNode(ISD::FMUL, SL, MVT::f32, LHSExt, Rcp, Op->getFlags());
13953 SDValue Err = DAG.getNode(FMADOpCode, SL, MVT::f32, NegRHSExt, Quot, LHSExt,
13954 Op->getFlags());
13955 Quot = DAG.getNode(FMADOpCode, SL, MVT::f32, Err, Rcp, Quot, Op->getFlags());
13956 Err = DAG.getNode(FMADOpCode, SL, MVT::f32, NegRHSExt, Quot, LHSExt,
13957 Op->getFlags());
13958 SDValue Tmp = DAG.getNode(ISD::FMUL, SL, MVT::f32, Err, Rcp, Op->getFlags());
13959 SDValue TmpCast = DAG.getNode(ISD::BITCAST, SL, MVT::i32, Tmp);
13960 TmpCast = DAG.getNode(ISD::AND, SL, MVT::i32, TmpCast,
13961 DAG.getConstant(0xff800000, SL, MVT::i32));
13962 Tmp = DAG.getNode(ISD::BITCAST, SL, MVT::f32, TmpCast);
13963 Quot = DAG.getNode(ISD::FADD, SL, MVT::f32, Tmp, Quot, Op->getFlags());
13964 SDValue RDst = DAG.getNode(ISD::FP_ROUND, SL, MVT::f16, Quot,
13965 DAG.getTargetConstant(0, SL, MVT::i32));
13966 return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f16, RDst, RHS, LHS,
13967 Op->getFlags());
13968}
13969
13970// Faster 2.5 ULP division that does not support denormals.
13971SDValue SITargetLowering::lowerFDIV_FAST(SDValue Op, SelectionDAG &DAG) const {
13972 SDNodeFlags Flags = Op->getFlags();
13973 SDLoc SL(Op);
13974 SDValue LHS = Op.getOperand(1);
13975 SDValue RHS = Op.getOperand(2);
13976
13977 // TODO: The combiner should probably handle elimination of redundant fabs.
13978 SDValue r1 = DAG.SignBitIsZeroFP(RHS)
13979 ? RHS
13980 : DAG.getNode(ISD::FABS, SL, MVT::f32, RHS, Flags);
13981
13982 const APFloat K0Val(0x1p+96f);
13983 const SDValue K0 = DAG.getConstantFP(K0Val, SL, MVT::f32);
13984
13985 const APFloat K1Val(0x1p-32f);
13986 const SDValue K1 = DAG.getConstantFP(K1Val, SL, MVT::f32);
13987
13988 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f32);
13989
13990 EVT SetCCVT =
13991 getSetCCResultType(DAG.getDataLayout(), *DAG.getContext(), MVT::f32);
13992
13993 SDValue r2 = DAG.getSetCC(SL, SetCCVT, r1, K0, ISD::SETOGT);
13994
13995 SDValue r3 = DAG.getNode(ISD::SELECT, SL, MVT::f32, r2, K1, One, Flags);
13996
13997 r1 = DAG.getNode(ISD::FMUL, SL, MVT::f32, RHS, r3, Flags);
13998
13999 // rcp does not support denormals.
14000 SDValue r0 = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32, r1, Flags);
14001
14002 SDValue Mul = DAG.getNode(ISD::FMUL, SL, MVT::f32, LHS, r0, Flags);
14003
14004 return DAG.getNode(ISD::FMUL, SL, MVT::f32, r3, Mul, Flags);
14005}
14006
14007// Returns immediate value for setting the F32 denorm mode when using the
14008// S_DENORM_MODE instruction.
14010 const SIMachineFunctionInfo *Info,
14011 const GCNSubtarget *ST) {
14012 assert(ST->hasDenormModeInst() && "Requires S_DENORM_MODE");
14013 uint32_t DPDenormModeDefault = Info->getMode().fpDenormModeDPValue();
14014 uint32_t Mode = SPDenormMode | (DPDenormModeDefault << 2);
14015 return DAG.getTargetConstant(Mode, SDLoc(), MVT::i32);
14016}
14017
14018SDValue SITargetLowering::LowerFDIV32(SDValue Op, SelectionDAG &DAG) const {
14019 if (SDValue FastLowered = lowerFastUnsafeFDIV(Op, DAG))
14020 return FastLowered;
14021
14022 // The selection matcher assumes anything with a chain selecting to a
14023 // mayRaiseFPException machine instruction. Since we're introducing a chain
14024 // here, we need to explicitly report nofpexcept for the regular fdiv
14025 // lowering.
14026 SDNodeFlags Flags = Op->getFlags();
14027 Flags.setNoFPExcept(true);
14028
14029 SDLoc SL(Op);
14030 SDValue LHS = Op.getOperand(0);
14031 SDValue RHS = Op.getOperand(1);
14032
14033 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f32);
14034
14035 SDVTList ScaleVT = DAG.getVTList(MVT::f32, MVT::i1);
14036
14037 SDValue DenominatorScaled =
14038 DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, {RHS, RHS, LHS}, Flags);
14039 SDValue NumeratorScaled =
14040 DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, {LHS, RHS, LHS}, Flags);
14041
14042 // Denominator is scaled to not be denormal, so using rcp is ok.
14043 SDValue ApproxRcp =
14044 DAG.getNode(AMDGPUISD::RCP, SL, MVT::f32, DenominatorScaled, Flags);
14045 SDValue NegDivScale0 =
14046 DAG.getNode(ISD::FNEG, SL, MVT::f32, DenominatorScaled, Flags);
14047
14048 using namespace AMDGPU::Hwreg;
14049 const unsigned Denorm32Reg = HwregEncoding::encode(ID_MODE, 4, 2);
14050 const SDValue BitField = DAG.getTargetConstant(Denorm32Reg, SL, MVT::i32);
14051
14052 const MachineFunction &MF = DAG.getMachineFunction();
14053 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
14054 const DenormalMode DenormMode = Info->getMode().FP32Denormals;
14055
14056 const bool PreservesDenormals = DenormMode == DenormalMode::getIEEE();
14057 const bool HasDynamicDenormals =
14058 (DenormMode.Input == DenormalMode::Dynamic) ||
14059 (DenormMode.Output == DenormalMode::Dynamic);
14060
14061 SDValue SavedDenormMode;
14062
14063 if (!PreservesDenormals) {
14064 // Note we can't use the STRICT_FMA/STRICT_FMUL for the non-strict FDIV
14065 // lowering. The chain dependence is insufficient, and we need glue. We do
14066 // not need the glue variants in a strictfp function.
14067
14068 SDVTList BindParamVTs = DAG.getVTList(MVT::Other, MVT::Glue);
14069
14070 SDValue Glue = DAG.getEntryNode();
14071 if (HasDynamicDenormals) {
14072 SDNode *GetReg = DAG.getMachineNode(AMDGPU::S_GETREG_B32, SL,
14073 DAG.getVTList(MVT::i32, MVT::Glue),
14074 {BitField, Glue});
14075 SavedDenormMode = SDValue(GetReg, 0);
14076
14077 Glue = DAG.getMergeValues(
14078 {DAG.getEntryNode(), SDValue(GetReg, 0), SDValue(GetReg, 1)}, SL);
14079 }
14080
14081 SDNode *EnableDenorm;
14082 if (Subtarget->hasDenormModeInst()) {
14083 const SDValue EnableDenormValue =
14084 getSPDenormModeValue(FP_DENORM_FLUSH_NONE, DAG, Info, Subtarget);
14085
14086 EnableDenorm = DAG.getNode(AMDGPUISD::DENORM_MODE, SL, BindParamVTs, Glue,
14087 EnableDenormValue)
14088 .getNode();
14089 } else {
14090 const SDValue EnableDenormValue =
14091 DAG.getConstant(FP_DENORM_FLUSH_NONE, SL, MVT::i32);
14092 EnableDenorm = DAG.getMachineNode(AMDGPU::S_SETREG_B32, SL, BindParamVTs,
14093 {EnableDenormValue, BitField, Glue});
14094 }
14095
14096 SDValue Ops[3] = {NegDivScale0, SDValue(EnableDenorm, 0),
14097 SDValue(EnableDenorm, 1)};
14098
14099 NegDivScale0 = DAG.getMergeValues(Ops, SL);
14100 }
14101
14102 SDValue Fma0 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0,
14103 ApproxRcp, One, NegDivScale0, Flags);
14104
14105 SDValue Fma1 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, Fma0, ApproxRcp,
14106 ApproxRcp, Fma0, Flags);
14107
14108 SDValue Mul = getFPBinOp(DAG, ISD::FMUL, SL, MVT::f32, NumeratorScaled, Fma1,
14109 Fma1, Flags);
14110
14111 SDValue Fma2 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0, Mul,
14112 NumeratorScaled, Mul, Flags);
14113
14114 SDValue Fma3 =
14115 getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, Fma2, Fma1, Mul, Fma2, Flags);
14116
14117 SDValue Fma4 = getFPTernOp(DAG, ISD::FMA, SL, MVT::f32, NegDivScale0, Fma3,
14118 NumeratorScaled, Fma3, Flags);
14119
14120 if (!PreservesDenormals) {
14121 SDNode *DisableDenorm;
14122 if (!HasDynamicDenormals && Subtarget->hasDenormModeInst()) {
14123 const SDValue DisableDenormValue = getSPDenormModeValue(
14124 FP_DENORM_FLUSH_IN_FLUSH_OUT, DAG, Info, Subtarget);
14125
14126 SDVTList BindParamVTs = DAG.getVTList(MVT::Other, MVT::Glue);
14127 DisableDenorm =
14128 DAG.getNode(AMDGPUISD::DENORM_MODE, SL, BindParamVTs,
14129 Fma4.getValue(1), DisableDenormValue, Fma4.getValue(2))
14130 .getNode();
14131 } else {
14132 assert(HasDynamicDenormals == (bool)SavedDenormMode);
14133 const SDValue DisableDenormValue =
14134 HasDynamicDenormals
14135 ? SavedDenormMode
14136 : DAG.getConstant(FP_DENORM_FLUSH_IN_FLUSH_OUT, SL, MVT::i32);
14137
14138 DisableDenorm = DAG.getMachineNode(
14139 AMDGPU::S_SETREG_B32, SL, MVT::Other,
14140 {DisableDenormValue, BitField, Fma4.getValue(1), Fma4.getValue(2)});
14141 }
14142
14143 SDValue OutputChain = DAG.getNode(ISD::TokenFactor, SL, MVT::Other,
14144 SDValue(DisableDenorm, 0), DAG.getRoot());
14145 DAG.setRoot(OutputChain);
14146 }
14147
14148 SDValue Scale = NumeratorScaled.getValue(1);
14149 SDValue Fmas = DAG.getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f32,
14150 {Fma4, Fma1, Fma3, Scale}, Flags);
14151
14152 return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f32, Fmas, RHS, LHS, Flags);
14153}
14154
14155SDValue SITargetLowering::LowerFDIV64(SDValue Op, SelectionDAG &DAG) const {
14156 if (SDValue FastLowered = lowerFastUnsafeFDIV64(Op, DAG))
14157 return FastLowered;
14158
14159 SDLoc SL(Op);
14160 SDValue X = Op.getOperand(0);
14161 SDValue Y = Op.getOperand(1);
14162
14163 const SDValue One = DAG.getConstantFP(1.0, SL, MVT::f64);
14164
14165 SDVTList ScaleVT = DAG.getVTList(MVT::f64, MVT::i1);
14166
14167 SDValue DivScale0 = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, Y, Y, X);
14168
14169 SDValue NegDivScale0 = DAG.getNode(ISD::FNEG, SL, MVT::f64, DivScale0);
14170
14171 SDValue Rcp = DAG.getNode(AMDGPUISD::RCP, SL, MVT::f64, DivScale0);
14172
14173 SDValue Fma0 = DAG.getNode(ISD::FMA, SL, MVT::f64, NegDivScale0, Rcp, One);
14174
14175 SDValue Fma1 = DAG.getNode(ISD::FMA, SL, MVT::f64, Rcp, Fma0, Rcp);
14176
14177 SDValue Fma2 = DAG.getNode(ISD::FMA, SL, MVT::f64, NegDivScale0, Fma1, One);
14178
14179 SDValue DivScale1 = DAG.getNode(AMDGPUISD::DIV_SCALE, SL, ScaleVT, X, Y, X);
14180
14181 SDValue Fma3 = DAG.getNode(ISD::FMA, SL, MVT::f64, Fma1, Fma2, Fma1);
14182 SDValue Mul = DAG.getNode(ISD::FMUL, SL, MVT::f64, DivScale1, Fma3);
14183
14184 SDValue Fma4 =
14185 DAG.getNode(ISD::FMA, SL, MVT::f64, NegDivScale0, Mul, DivScale1);
14186
14187 SDValue Scale;
14188
14189 if (!Subtarget->hasUsableDivScaleConditionOutput()) {
14190 // Workaround a hardware bug on SI where the condition output from div_scale
14191 // is not usable.
14192
14193 const SDValue Hi = DAG.getConstant(1, SL, MVT::i32);
14194
14195 // Figure out if the scale to use for div_fmas.
14196 SDValue NumBC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, X);
14197 SDValue DenBC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, Y);
14198 SDValue Scale0BC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, DivScale0);
14199 SDValue Scale1BC = DAG.getNode(ISD::BITCAST, SL, MVT::v2i32, DivScale1);
14200
14201 SDValue NumHi =
14202 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, NumBC, Hi);
14203 SDValue DenHi =
14204 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, DenBC, Hi);
14205
14206 SDValue Scale0Hi =
14207 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Scale0BC, Hi);
14208 SDValue Scale1Hi =
14209 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Scale1BC, Hi);
14210
14211 SDValue CmpDen = DAG.getSetCC(SL, MVT::i1, DenHi, Scale0Hi, ISD::SETEQ);
14212 SDValue CmpNum = DAG.getSetCC(SL, MVT::i1, NumHi, Scale1Hi, ISD::SETEQ);
14213 Scale = DAG.getNode(ISD::XOR, SL, MVT::i1, CmpNum, CmpDen);
14214 } else {
14215 Scale = DivScale1.getValue(1);
14216 }
14217
14218 SDValue Fmas =
14219 DAG.getNode(AMDGPUISD::DIV_FMAS, SL, MVT::f64, Fma4, Fma3, Mul, Scale);
14220
14221 return DAG.getNode(AMDGPUISD::DIV_FIXUP, SL, MVT::f64, Fmas, Y, X);
14222}
14223
14224SDValue SITargetLowering::LowerFDIV(SDValue Op, SelectionDAG &DAG) const {
14225 EVT VT = Op.getValueType();
14226
14227 if (VT == MVT::f32)
14228 return LowerFDIV32(Op, DAG);
14229
14230 if (VT == MVT::f64)
14231 return LowerFDIV64(Op, DAG);
14232
14233 if (VT == MVT::f16 || VT == MVT::bf16)
14234 return LowerFDIV16(Op, DAG);
14235
14236 llvm_unreachable("Unexpected type for fdiv");
14237}
14238
14239SDValue SITargetLowering::LowerFFREXP(SDValue Op, SelectionDAG &DAG) const {
14240 SDLoc dl(Op);
14241 SDValue Val = Op.getOperand(0);
14242 EVT VT = Val.getValueType();
14243 EVT ResultExpVT = Op->getValueType(1);
14244 EVT InstrExpVT = VT == MVT::f16 ? MVT::i16 : MVT::i32;
14245
14246 SDValue Mant = DAG.getNode(
14248 DAG.getTargetConstant(Intrinsic::amdgcn_frexp_mant, dl, MVT::i32), Val);
14249
14250 SDValue Exp = DAG.getNode(
14251 ISD::INTRINSIC_WO_CHAIN, dl, InstrExpVT,
14252 DAG.getTargetConstant(Intrinsic::amdgcn_frexp_exp, dl, MVT::i32), Val);
14253
14254 if (Subtarget->hasFractBug()) {
14255 SDValue Fabs = DAG.getNode(ISD::FABS, dl, VT, Val);
14256 SDValue Inf =
14258
14259 SDValue IsFinite = DAG.getSetCC(dl, MVT::i1, Fabs, Inf, ISD::SETOLT);
14260 SDValue Zero = DAG.getConstant(0, dl, InstrExpVT);
14261 Exp = DAG.getNode(ISD::SELECT, dl, InstrExpVT, IsFinite, Exp, Zero);
14262 Mant = DAG.getNode(ISD::SELECT, dl, VT, IsFinite, Mant, Val);
14263 }
14264
14265 SDValue CastExp = DAG.getSExtOrTrunc(Exp, dl, ResultExpVT);
14266 return DAG.getMergeValues({Mant, CastExp}, dl);
14267}
14268
14269SDValue SITargetLowering::LowerSTORE(SDValue Op, SelectionDAG &DAG) const {
14270 SDLoc DL(Op);
14271 StoreSDNode *Store = cast<StoreSDNode>(Op);
14272 EVT VT = Store->getMemoryVT();
14273
14274 if (VT == MVT::i1) {
14275 return DAG.getTruncStore(
14276 Store->getChain(), DL,
14277 DAG.getSExtOrTrunc(Store->getValue(), DL, MVT::i32),
14278 Store->getBasePtr(), MVT::i1, Store->getMemOperand());
14279 }
14280
14281 assert(VT.isVector() &&
14282 Store->getValue().getValueType().getScalarType() == MVT::i32);
14283
14284 unsigned AS = Store->getAddressSpace();
14285 if (Subtarget->hasLDSMisalignedBugInWGPMode() &&
14286 AS == AMDGPUAS::FLAT_ADDRESS &&
14287 Store->getAlign().value() < VT.getStoreSize() &&
14288 VT.getSizeInBits() > 32) {
14289 return SplitVectorStore(Op, DAG);
14290 }
14291
14293 SIMachineFunctionInfo *MFI = MF.getInfo<SIMachineFunctionInfo>();
14294 // If there is a possibility that flat instruction access scratch memory
14295 // then we need to use the same legalization rules we use for private.
14296 if (AS == AMDGPUAS::FLAT_ADDRESS &&
14297 !Subtarget->hasMultiDwordFlatScratchAddressing())
14298 AS = addressMayBeAccessedAsPrivate(Store->getMemOperand(), *MFI)
14301
14302 unsigned NumElements = VT.getVectorNumElements();
14304 if (NumElements > 4)
14305 return SplitVectorStore(Op, DAG);
14306 // v3 stores not supported on SI.
14307 if (NumElements == 3 && !Subtarget->hasDwordx3LoadStores())
14308 return SplitVectorStore(Op, DAG);
14309
14311 VT, *Store->getMemOperand()))
14312 return expandUnalignedStore(Store, DAG);
14313
14314 return SDValue();
14315 }
14316 if (AS == AMDGPUAS::PRIVATE_ADDRESS) {
14317 switch (Subtarget->getMaxPrivateElementSize()) {
14318 case 4:
14319 return scalarizeVectorStore(Store, DAG);
14320 case 8:
14321 if (NumElements > 2)
14322 return SplitVectorStore(Op, DAG);
14323 return SDValue();
14324 case 16:
14325 if (NumElements > 4 ||
14326 (NumElements == 3 && !Subtarget->hasFlatScratchEnabled()))
14327 return SplitVectorStore(Op, DAG);
14328 return SDValue();
14329 default:
14330 llvm_unreachable("unsupported private_element_size");
14331 }
14332 } else if (AS == AMDGPUAS::LOCAL_ADDRESS || AS == AMDGPUAS::REGION_ADDRESS) {
14333 unsigned Fast = 0;
14334 auto Flags = Store->getMemOperand()->getFlags();
14336 Store->getAlign(), Flags, &Fast) &&
14337 Fast > 1)
14338 return SDValue();
14339
14340 if (VT.isVector())
14341 return SplitVectorStore(Op, DAG);
14342
14343 return expandUnalignedStore(Store, DAG);
14344 }
14345
14346 // Probably an invalid store. If so we'll end up emitting a selection error.
14347 return SDValue();
14348}
14349
14350// Avoid the full correct expansion for f32 sqrt when promoting from f16.
14351SDValue SITargetLowering::lowerFSQRTF16(SDValue Op, SelectionDAG &DAG) const {
14352 SDLoc SL(Op);
14353 assert(!Subtarget->has16BitInsts());
14354 SDNodeFlags Flags = Op->getFlags();
14355 SDValue Ext =
14356 DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, Op.getOperand(0), Flags);
14357
14358 SDValue SqrtID = DAG.getTargetConstant(Intrinsic::amdgcn_sqrt, SL, MVT::i32);
14359 SDValue Sqrt =
14360 DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::f32, SqrtID, Ext, Flags);
14361
14362 return DAG.getNode(ISD::FP_ROUND, SL, MVT::f16, Sqrt,
14363 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
14364}
14365
14366SDValue SITargetLowering::lowerFSQRTF32(SDValue Op, SelectionDAG &DAG) const {
14367 SDLoc DL(Op);
14368 SDNodeFlags Flags = Op->getFlags();
14369 MVT VT = Op.getValueType().getSimpleVT();
14370 const SDValue X = Op.getOperand(0);
14371
14372 if (allowApproxFunc(DAG, Flags)) {
14373 // Instruction is 1ulp but ignores denormals.
14374 return DAG.getNode(
14376 DAG.getTargetConstant(Intrinsic::amdgcn_sqrt, DL, MVT::i32), X, Flags);
14377 }
14378
14379 SDValue ScaleThreshold = DAG.getConstantFP(0x1.0p-96f, DL, VT);
14380 SDValue NeedScale = DAG.getSetCC(DL, MVT::i1, X, ScaleThreshold, ISD::SETOLT);
14381
14382 SDValue ScaleUpFactor = DAG.getConstantFP(0x1.0p+32f, DL, VT);
14383
14384 SDValue ScaledX = DAG.getNode(ISD::FMUL, DL, VT, X, ScaleUpFactor, Flags);
14385
14386 SDValue SqrtX =
14387 DAG.getNode(ISD::SELECT, DL, VT, NeedScale, ScaledX, X, Flags);
14388
14389 SDValue SqrtS;
14390 if (needsDenormHandlingF32(DAG, X, Flags)) {
14391 SDValue SqrtID =
14392 DAG.getTargetConstant(Intrinsic::amdgcn_sqrt, DL, MVT::i32);
14393 SqrtS = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, DL, VT, SqrtID, SqrtX, Flags);
14394
14395 SDValue SqrtSAsInt = DAG.getNode(ISD::BITCAST, DL, MVT::i32, SqrtS);
14396 SDValue SqrtSNextDownInt =
14397 DAG.getNode(ISD::ADD, DL, MVT::i32, SqrtSAsInt,
14398 DAG.getAllOnesConstant(DL, MVT::i32));
14399 SDValue SqrtSNextDown = DAG.getNode(ISD::BITCAST, DL, VT, SqrtSNextDownInt);
14400
14401 SDValue NegSqrtSNextDown =
14402 DAG.getNode(ISD::FNEG, DL, VT, SqrtSNextDown, Flags);
14403
14404 SDValue SqrtVP =
14405 DAG.getNode(ISD::FMA, DL, VT, NegSqrtSNextDown, SqrtS, SqrtX, Flags);
14406
14407 SDValue SqrtSNextUpInt = DAG.getNode(ISD::ADD, DL, MVT::i32, SqrtSAsInt,
14408 DAG.getConstant(1, DL, MVT::i32));
14409 SDValue SqrtSNextUp = DAG.getNode(ISD::BITCAST, DL, VT, SqrtSNextUpInt);
14410
14411 SDValue NegSqrtSNextUp = DAG.getNode(ISD::FNEG, DL, VT, SqrtSNextUp, Flags);
14412 SDValue SqrtVS =
14413 DAG.getNode(ISD::FMA, DL, VT, NegSqrtSNextUp, SqrtS, SqrtX, Flags);
14414
14415 SDValue Zero = DAG.getConstantFP(0.0f, DL, VT);
14416 SDValue SqrtVPLE0 = DAG.getSetCC(DL, MVT::i1, SqrtVP, Zero, ISD::SETOLE);
14417
14418 SqrtS = DAG.getNode(ISD::SELECT, DL, VT, SqrtVPLE0, SqrtSNextDown, SqrtS,
14419 Flags);
14420
14421 SDValue SqrtVPVSGT0 = DAG.getSetCC(DL, MVT::i1, SqrtVS, Zero, ISD::SETOGT);
14422 SqrtS = DAG.getNode(ISD::SELECT, DL, VT, SqrtVPVSGT0, SqrtSNextUp, SqrtS,
14423 Flags);
14424 } else {
14425 SDValue SqrtR = DAG.getNode(AMDGPUISD::RSQ, DL, VT, SqrtX, Flags);
14426
14427 SqrtS = DAG.getNode(ISD::FMUL, DL, VT, SqrtX, SqrtR, Flags);
14428
14429 SDValue Half = DAG.getConstantFP(0.5f, DL, VT);
14430 SDValue SqrtH = DAG.getNode(ISD::FMUL, DL, VT, SqrtR, Half, Flags);
14431 SDValue NegSqrtH = DAG.getNode(ISD::FNEG, DL, VT, SqrtH, Flags);
14432
14433 SDValue SqrtE = DAG.getNode(ISD::FMA, DL, VT, NegSqrtH, SqrtS, Half, Flags);
14434 SqrtH = DAG.getNode(ISD::FMA, DL, VT, SqrtH, SqrtE, SqrtH, Flags);
14435 SqrtS = DAG.getNode(ISD::FMA, DL, VT, SqrtS, SqrtE, SqrtS, Flags);
14436
14437 SDValue NegSqrtS = DAG.getNode(ISD::FNEG, DL, VT, SqrtS, Flags);
14438 SDValue SqrtD =
14439 DAG.getNode(ISD::FMA, DL, VT, NegSqrtS, SqrtS, SqrtX, Flags);
14440 SqrtS = DAG.getNode(ISD::FMA, DL, VT, SqrtD, SqrtH, SqrtS, Flags);
14441 }
14442
14443 SDValue ScaleDownFactor = DAG.getConstantFP(0x1.0p-16f, DL, VT);
14444
14445 SDValue ScaledDown =
14446 DAG.getNode(ISD::FMUL, DL, VT, SqrtS, ScaleDownFactor, Flags);
14447
14448 SqrtS = DAG.getNode(ISD::SELECT, DL, VT, NeedScale, ScaledDown, SqrtS, Flags);
14449 SDValue IsZeroOrInf =
14450 DAG.getNode(ISD::IS_FPCLASS, DL, MVT::i1, SqrtX,
14451 DAG.getTargetConstant(fcZero | fcPosInf, DL, MVT::i32));
14452
14453 return DAG.getNode(ISD::SELECT, DL, VT, IsZeroOrInf, SqrtX, SqrtS, Flags);
14454}
14455
14456SDValue SITargetLowering::lowerFSQRTF64(SDValue Op, SelectionDAG &DAG) const {
14457 // For double type, the SQRT and RSQ instructions don't have required
14458 // precision, we apply Goldschmidt's algorithm to improve the result:
14459 //
14460 // y0 = rsq(x)
14461 // g0 = x * y0
14462 // h0 = 0.5 * y0
14463 //
14464 // r0 = 0.5 - h0 * g0
14465 // g1 = g0 * r0 + g0
14466 // h1 = h0 * r0 + h0
14467 //
14468 // r1 = 0.5 - h1 * g1 => d0 = x - g1 * g1
14469 // g2 = g1 * r1 + g1 g2 = d0 * h1 + g1
14470 // h2 = h1 * r1 + h1
14471 //
14472 // r2 = 0.5 - h2 * g2 => d1 = x - g2 * g2
14473 // g3 = g2 * r2 + g2 g3 = d1 * h1 + g2
14474 //
14475 // sqrt(x) = g3
14476
14477 SDNodeFlags Flags = Op->getFlags();
14478
14479 SDLoc DL(Op);
14480
14481 SDValue X = Op.getOperand(0);
14482 SDValue ZeroInt = DAG.getConstant(0, DL, MVT::i32);
14483
14484 SDValue SqrtX = X;
14485 SDValue Scaling;
14486 if (!Flags.hasApproximateFuncs()) {
14487 SDValue ScaleConstant = DAG.getConstantFP(0x1.0p-767, DL, MVT::f64);
14488 Scaling = DAG.getSetCC(DL, MVT::i1, X, ScaleConstant, ISD::SETOLT);
14489
14490 // Scale up input if it is too small.
14491 SDValue ScaleUpFactor = DAG.getConstant(256, DL, MVT::i32);
14492 SDValue ScaleUp =
14493 DAG.getNode(ISD::SELECT, DL, MVT::i32, Scaling, ScaleUpFactor, ZeroInt);
14494 SqrtX = DAG.getNode(ISD::FLDEXP, DL, MVT::f64, X, ScaleUp, Flags);
14495 }
14496
14497 SDValue SqrtY = DAG.getNode(AMDGPUISD::RSQ, DL, MVT::f64, SqrtX);
14498
14499 SDValue SqrtS0 = DAG.getNode(ISD::FMUL, DL, MVT::f64, SqrtX, SqrtY);
14500
14501 SDValue Half = DAG.getConstantFP(0.5, DL, MVT::f64);
14502 SDValue SqrtH0 = DAG.getNode(ISD::FMUL, DL, MVT::f64, SqrtY, Half);
14503
14504 SDValue NegSqrtH0 = DAG.getNode(ISD::FNEG, DL, MVT::f64, SqrtH0);
14505 SDValue SqrtR0 = DAG.getNode(ISD::FMA, DL, MVT::f64, NegSqrtH0, SqrtS0, Half);
14506
14507 SDValue SqrtH1 = DAG.getNode(ISD::FMA, DL, MVT::f64, SqrtH0, SqrtR0, SqrtH0);
14508
14509 SDValue SqrtS1 = DAG.getNode(ISD::FMA, DL, MVT::f64, SqrtS0, SqrtR0, SqrtS0);
14510
14511 SDValue NegSqrtS1 = DAG.getNode(ISD::FNEG, DL, MVT::f64, SqrtS1);
14512 SDValue SqrtD0 =
14513 DAG.getNode(ISD::FMA, DL, MVT::f64, NegSqrtS1, SqrtS1, SqrtX);
14514
14515 SDValue SqrtS2 = DAG.getNode(ISD::FMA, DL, MVT::f64, SqrtD0, SqrtH1, SqrtS1);
14516
14517 SDValue SqrtRet = SqrtS2;
14518 if (!Flags.hasApproximateFuncs()) {
14519 SDValue NegSqrtS2 = DAG.getNode(ISD::FNEG, DL, MVT::f64, SqrtS2);
14520 SDValue SqrtD1 =
14521 DAG.getNode(ISD::FMA, DL, MVT::f64, NegSqrtS2, SqrtS2, SqrtX);
14522
14523 SqrtRet = DAG.getNode(ISD::FMA, DL, MVT::f64, SqrtD1, SqrtH1, SqrtS2);
14524
14525 SDValue ScaleDownFactor = DAG.getSignedConstant(-128, DL, MVT::i32);
14526 SDValue ScaleDown = DAG.getNode(ISD::SELECT, DL, MVT::i32, Scaling,
14527 ScaleDownFactor, ZeroInt);
14528 SqrtRet = DAG.getNode(ISD::FLDEXP, DL, MVT::f64, SqrtRet, ScaleDown, Flags);
14529 }
14530
14531 // TODO: Check for DAZ and expand to subnormals
14532
14533 SDValue IsZeroOrInf;
14534 if (Flags.hasNoInfs()) {
14535 SDValue Zero = DAG.getConstantFP(0.0, DL, MVT::f64);
14536 IsZeroOrInf = DAG.getSetCC(DL, MVT::i1, SqrtX, Zero, ISD::SETOEQ);
14537 } else {
14538 IsZeroOrInf =
14539 DAG.getNode(ISD::IS_FPCLASS, DL, MVT::i1, SqrtX,
14540 DAG.getTargetConstant(fcZero | fcPosInf, DL, MVT::i32));
14541 }
14542
14543 // If x is +INF, +0, or -0, use its original value
14544 return DAG.getNode(ISD::SELECT, DL, MVT::f64, IsZeroOrInf, SqrtX, SqrtRet,
14545 Flags);
14546}
14547
14548SDValue SITargetLowering::LowerTrig(SDValue Op, SelectionDAG &DAG) const {
14549 SDLoc DL(Op);
14550 EVT VT = Op.getValueType();
14551 SDValue Arg = Op.getOperand(0);
14552 SDValue TrigVal;
14553
14554 // Propagate fast-math flags so that the multiply we introduce can be folded
14555 // if Arg is already the result of a multiply by constant.
14556 auto Flags = Op->getFlags();
14557
14558 // AMDGPUISD nodes of vector type must be unrolled here since
14559 // they will not be expanded elsewhere.
14560 auto UnrollIfVec = [&DAG](SDValue V) -> SDValue {
14561 if (!V.getValueType().isVector())
14562 return V;
14563
14564 return DAG.UnrollVectorOp(cast<SDNode>(V));
14565 };
14566
14567 SDValue OneOver2Pi = DAG.getConstantFP(0.5 * numbers::inv_pi, DL, VT);
14568
14569 if (Subtarget->hasTrigReducedRange()) {
14570 SDValue MulVal = DAG.getNode(ISD::FMUL, DL, VT, Arg, OneOver2Pi, Flags);
14571 TrigVal = UnrollIfVec(DAG.getNode(AMDGPUISD::FRACT, DL, VT, MulVal, Flags));
14572 } else {
14573 TrigVal = DAG.getNode(ISD::FMUL, DL, VT, Arg, OneOver2Pi, Flags);
14574 }
14575
14576 switch (Op.getOpcode()) {
14577 case ISD::FCOS:
14578 TrigVal = DAG.getNode(AMDGPUISD::COS_HW, SDLoc(Op), VT, TrigVal, Flags);
14579 break;
14580 case ISD::FSIN:
14581 TrigVal = DAG.getNode(AMDGPUISD::SIN_HW, SDLoc(Op), VT, TrigVal, Flags);
14582 break;
14583 default:
14584 llvm_unreachable("Wrong trig opcode");
14585 }
14586
14587 return UnrollIfVec(TrigVal);
14588}
14589
14590SDValue SITargetLowering::LowerATOMIC_CMP_SWAP(SDValue Op,
14591 SelectionDAG &DAG) const {
14592 AtomicSDNode *AtomicNode = cast<AtomicSDNode>(Op);
14593 assert(AtomicNode->isCompareAndSwap());
14594 unsigned AS = AtomicNode->getAddressSpace();
14595
14596 // No custom lowering required for local address space
14598 return Op;
14599
14600 // Non-local address space requires custom lowering for atomic compare
14601 // and swap; cmp and swap should be in a v2i32 or v2i64 in case of _X2
14602 SDLoc DL(Op);
14603 SDValue ChainIn = Op.getOperand(0);
14604 SDValue Addr = Op.getOperand(1);
14605 SDValue Old = Op.getOperand(2);
14606 SDValue New = Op.getOperand(3);
14607 EVT VT = Op.getValueType();
14608 MVT SimpleVT = VT.getSimpleVT();
14609 MVT VecType = MVT::getVectorVT(SimpleVT, 2);
14610
14611 SDValue NewOld = DAG.getBuildVector(VecType, DL, {New, Old});
14612 SDValue Ops[] = {ChainIn, Addr, NewOld};
14613
14614 return DAG.getMemIntrinsicNode(AMDGPUISD::ATOMIC_CMP_SWAP, DL,
14615 Op->getVTList(), Ops, VT,
14616 AtomicNode->getMemOperand());
14617}
14618
14619//===----------------------------------------------------------------------===//
14620// Custom DAG optimizations
14621//===----------------------------------------------------------------------===//
14622
14623SDValue
14624SITargetLowering::performUCharToFloatCombine(SDNode *N,
14625 DAGCombinerInfo &DCI) const {
14626 EVT VT = N->getValueType(0);
14627 EVT ScalarVT = VT.getScalarType();
14628 if (ScalarVT != MVT::f32 && ScalarVT != MVT::f16)
14629 return SDValue();
14630
14631 SelectionDAG &DAG = DCI.DAG;
14632 SDLoc DL(N);
14633
14634 SDValue Src = N->getOperand(0);
14635 EVT SrcVT = Src.getValueType();
14636
14637 // TODO: We could try to match extracting the higher bytes, which would be
14638 // easier if i8 vectors weren't promoted to i32 vectors, particularly after
14639 // types are legalized. v4i8 -> v4f32 is probably the only case to worry
14640 // about in practice.
14641 if (DCI.isAfterLegalizeDAG() && SrcVT == MVT::i32) {
14642 if (DAG.MaskedValueIsZero(Src, APInt::getHighBitsSet(32, 24))) {
14643 SDValue Cvt = DAG.getNode(AMDGPUISD::CVT_F32_UBYTE0, DL, MVT::f32, Src);
14644 DCI.AddToWorklist(Cvt.getNode());
14645
14646 // For the f16 case, fold to a cast to f32 and then cast back to f16.
14647 if (ScalarVT != MVT::f32) {
14648 Cvt = DAG.getNode(ISD::FP_ROUND, DL, VT, Cvt,
14649 DAG.getTargetConstant(0, DL, MVT::i32));
14650 }
14651 return Cvt;
14652 }
14653 }
14654
14655 return SDValue();
14656}
14657
14658SDValue SITargetLowering::performFCopySignCombine(SDNode *N,
14659 DAGCombinerInfo &DCI) const {
14660 SDValue MagnitudeOp = N->getOperand(0);
14661 SDValue SignOp = N->getOperand(1);
14662
14663 // The generic combine for fcopysign + fp cast is too conservative with
14664 // vectors, and also gets confused by the splitting we will perform here, so
14665 // peek through FP casts.
14666 if (SignOp.getOpcode() == ISD::FP_EXTEND ||
14667 SignOp.getOpcode() == ISD::FP_ROUND)
14668 SignOp = SignOp.getOperand(0);
14669
14670 SelectionDAG &DAG = DCI.DAG;
14671 SDLoc DL(N);
14672 EVT SignVT = SignOp.getValueType();
14673
14674 // f64 fcopysign is really an f32 copysign on the high bits, so replace the
14675 // lower half with a copy.
14676 // fcopysign f64:x, _:y -> x.lo32, (fcopysign (f32 x.hi32), _:y)
14677 EVT MagVT = MagnitudeOp.getValueType();
14678
14679 unsigned NumElts = MagVT.isVector() ? MagVT.getVectorNumElements() : 1;
14680
14681 if (MagVT.getScalarType() == MVT::f64) {
14682 EVT F32VT = MagVT.isVector()
14683 ? EVT::getVectorVT(*DAG.getContext(), MVT::f32, 2 * NumElts)
14684 : MVT::v2f32;
14685
14686 SDValue MagAsVector = DAG.getNode(ISD::BITCAST, DL, F32VT, MagnitudeOp);
14687
14689 for (unsigned I = 0; I != NumElts; ++I) {
14690 SDValue MagLo =
14691 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, MagAsVector,
14692 DAG.getConstant(2 * I, DL, MVT::i32));
14693 SDValue MagHi =
14694 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, MagAsVector,
14695 DAG.getConstant(2 * I + 1, DL, MVT::i32));
14696
14697 SDValue SignOpElt =
14698 MagVT.isVector()
14700 SignOp, DAG.getConstant(I, DL, MVT::i32))
14701 : SignOp;
14702
14703 SDValue HiOp =
14704 DAG.getNode(ISD::FCOPYSIGN, DL, MVT::f32, MagHi, SignOpElt);
14705
14706 SDValue Vector =
14707 DAG.getNode(ISD::BUILD_VECTOR, DL, MVT::v2f32, MagLo, HiOp);
14708
14709 SDValue NewElt = DAG.getNode(ISD::BITCAST, DL, MVT::f64, Vector);
14710 NewElts.push_back(NewElt);
14711 }
14712
14713 if (NewElts.size() == 1)
14714 return NewElts[0];
14715
14716 return DAG.getNode(ISD::BUILD_VECTOR, DL, MagVT, NewElts);
14717 }
14718
14719 if (SignVT.getScalarType() != MVT::f64)
14720 return SDValue();
14721
14722 // Reduce width of sign operand, we only need the highest bit.
14723 //
14724 // fcopysign f64:x, f64:y ->
14725 // fcopysign f64:x, (extract_vector_elt (bitcast f64:y to v2f32), 1)
14726 // TODO: In some cases it might make sense to go all the way to f16.
14727
14728 EVT F32VT = MagVT.isVector()
14729 ? EVT::getVectorVT(*DAG.getContext(), MVT::f32, 2 * NumElts)
14730 : MVT::v2f32;
14731
14732 SDValue SignAsVector = DAG.getNode(ISD::BITCAST, DL, F32VT, SignOp);
14733
14734 SmallVector<SDValue, 8> F32Signs;
14735 for (unsigned I = 0; I != NumElts; ++I) {
14736 // Take sign from odd elements of cast vector
14737 SDValue SignAsF32 =
14738 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, MVT::f32, SignAsVector,
14739 DAG.getConstant(2 * I + 1, DL, MVT::i32));
14740 F32Signs.push_back(SignAsF32);
14741 }
14742
14743 SDValue NewSign =
14744 NumElts == 1
14745 ? F32Signs.back()
14747 EVT::getVectorVT(*DAG.getContext(), MVT::f32, NumElts),
14748 F32Signs);
14749
14750 return DAG.getNode(ISD::FCOPYSIGN, DL, N->getValueType(0), N->getOperand(0),
14751 NewSign);
14752}
14753
14754// (shl (add x, c1), c2) -> add (shl x, c2), (shl c1, c2)
14755// (shl (or x, c1), c2) -> add (shl x, c2), (shl c1, c2) iff x and c1 share no
14756// bits
14757
14758// This is a variant of
14759// (mul (add x, c1), c2) -> add (mul x, c2), (mul c1, c2),
14760//
14761// The normal DAG combiner will do this, but only if the add has one use since
14762// that would increase the number of instructions.
14763//
14764// This prevents us from seeing a constant offset that can be folded into a
14765// memory instruction's addressing mode. If we know the resulting add offset of
14766// a pointer can be folded into an addressing offset, we can replace the pointer
14767// operand with the add of new constant offset. This eliminates one of the uses,
14768// and may allow the remaining use to also be simplified.
14769//
14770SDValue SITargetLowering::performSHLPtrCombine(SDNode *N, unsigned AddrSpace,
14771 EVT MemVT,
14772 DAGCombinerInfo &DCI) const {
14773 SDValue N0 = N->getOperand(0);
14774 SDValue N1 = N->getOperand(1);
14775
14776 // We only do this to handle cases where it's profitable when there are
14777 // multiple uses of the add, so defer to the standard combine.
14778 if ((!N0->isAnyAdd() && N0.getOpcode() != ISD::OR) || N0->hasOneUse())
14779 return SDValue();
14780
14781 const ConstantSDNode *CN1 = dyn_cast<ConstantSDNode>(N1);
14782 if (!CN1)
14783 return SDValue();
14784
14785 const ConstantSDNode *CAdd = dyn_cast<ConstantSDNode>(N0.getOperand(1));
14786 if (!CAdd)
14787 return SDValue();
14788
14789 SelectionDAG &DAG = DCI.DAG;
14790
14791 if (N0->getOpcode() == ISD::OR &&
14792 !DAG.haveNoCommonBitsSet(N0.getOperand(0), N0.getOperand(1)))
14793 return SDValue();
14794
14795 // If the resulting offset is too large, we can't fold it into the
14796 // addressing mode offset.
14797 APInt Offset = CAdd->getAPIntValue() << CN1->getAPIntValue();
14798 Type *Ty = MemVT.getTypeForEVT(*DCI.DAG.getContext());
14799
14800 AddrMode AM;
14801 AM.HasBaseReg = true;
14802 AM.BaseOffs = Offset.getSExtValue();
14803 if (!isLegalAddressingMode(DCI.DAG.getDataLayout(), AM, Ty, AddrSpace))
14804 return SDValue();
14805
14806 SDLoc SL(N);
14807 EVT VT = N->getValueType(0);
14808
14809 SDValue ShlX = DAG.getNode(ISD::SHL, SL, VT, N0.getOperand(0), N1);
14810 SDValue COffset = DAG.getConstant(Offset, SL, VT);
14811
14812 SDNodeFlags Flags;
14813 Flags.setNoUnsignedWrap(
14814 N->getFlags().hasNoUnsignedWrap() &&
14815 (N0.getOpcode() == ISD::OR || N0->getFlags().hasNoUnsignedWrap()));
14816
14817 // Use ISD::ADD even if the original operation was ISD::PTRADD, since we can't
14818 // be sure that the new left operand is a proper base pointer.
14819 return DAG.getNode(ISD::ADD, SL, VT, ShlX, COffset, Flags);
14820}
14821
14822/// MemSDNode::getBasePtr() does not work for intrinsics, which needs to offset
14823/// by the chain and intrinsic ID. Theoretically we would also need to check the
14824/// specific intrinsic, but they all place the pointer operand first.
14825static unsigned getBasePtrIndex(const MemSDNode *N) {
14826 switch (N->getOpcode()) {
14827 case ISD::STORE:
14830 return 2;
14831 default:
14832 return 1;
14833 }
14834}
14835
14836SDValue SITargetLowering::performMemSDNodeCombine(MemSDNode *N,
14837 DAGCombinerInfo &DCI) const {
14838 SelectionDAG &DAG = DCI.DAG;
14839
14840 unsigned PtrIdx = getBasePtrIndex(N);
14841 SDValue Ptr = N->getOperand(PtrIdx);
14842
14843 // TODO: We could also do this for multiplies.
14844 if (Ptr.getOpcode() == ISD::SHL) {
14845 SDValue NewPtr = performSHLPtrCombine(Ptr.getNode(), N->getAddressSpace(),
14846 N->getMemoryVT(), DCI);
14847 if (NewPtr) {
14848 SmallVector<SDValue, 8> NewOps(N->ops());
14849
14850 NewOps[PtrIdx] = NewPtr;
14851 return SDValue(DAG.UpdateNodeOperands(N, NewOps), 0);
14852 }
14853 }
14854
14855 return SDValue();
14856}
14857
14858static bool bitOpWithConstantIsReducible(unsigned Opc, uint32_t Val) {
14859 return (Opc == ISD::AND && (Val == 0 || Val == 0xffffffff)) ||
14860 (Opc == ISD::OR && (Val == 0xffffffff || Val == 0)) ||
14861 (Opc == ISD::XOR && Val == 0);
14862}
14863
14864// Break up 64-bit bit operation of a constant into two 32-bit and/or/xor. This
14865// will typically happen anyway for a VALU 64-bit and. This exposes other 32-bit
14866// integer combine opportunities since most 64-bit operations are decomposed
14867// this way. TODO: We won't want this for SALU especially if it is an inline
14868// immediate.
14869SDValue SITargetLowering::splitBinaryBitConstantOp(
14870 DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS,
14871 const ConstantSDNode *CRHS) const {
14872 uint64_t Val = CRHS->getZExtValue();
14873 uint32_t ValLo = Lo_32(Val);
14874 uint32_t ValHi = Hi_32(Val);
14875 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
14876
14877 if ((bitOpWithConstantIsReducible(Opc, ValLo) ||
14879 (CRHS->hasOneUse() && !TII->isInlineConstant(CRHS->getAPIntValue()))) {
14880 // We have 64-bit scalar and/or/xor, but do not have vector forms.
14881 if (Subtarget->has64BitLiterals() && CRHS->hasOneUse() &&
14882 !CRHS->user_begin()->isDivergent())
14883 return SDValue();
14884
14885 // If we need to materialize a 64-bit immediate, it will be split up later
14886 // anyway. Avoid creating the harder to understand 64-bit immediate
14887 // materialization.
14888 return splitBinaryBitConstantOpImpl(DCI, SL, Opc, LHS, ValLo, ValHi);
14889 }
14890
14891 return SDValue();
14892}
14893
14895 if (V.getValueType() != MVT::i1)
14896 return false;
14897 switch (V.getOpcode()) {
14898 default:
14899 break;
14900 case ISD::SETCC:
14901 case ISD::IS_FPCLASS:
14902 case AMDGPUISD::FP_CLASS:
14903 return true;
14904 case ISD::AND:
14905 case ISD::OR:
14906 case ISD::XOR:
14907 return isBoolSGPR(V.getOperand(0)) && isBoolSGPR(V.getOperand(1));
14908 case ISD::SADDO:
14909 case ISD::UADDO:
14910 case ISD::SSUBO:
14911 case ISD::USUBO:
14912 case ISD::SMULO:
14913 case ISD::UMULO:
14914 return V.getResNo() == 1;
14916 unsigned IntrinsicID = V.getConstantOperandVal(0);
14917 switch (IntrinsicID) {
14918 case Intrinsic::amdgcn_is_shared:
14919 case Intrinsic::amdgcn_is_private:
14920 return true;
14921 default:
14922 return false;
14923 }
14924
14925 return false;
14926 }
14927 }
14928 return false;
14929}
14930
14931// If a constant has all zeroes or all ones within each byte return it.
14932// Otherwise return 0.
14934 // 0xff for any zero byte in the mask
14935 uint32_t ZeroByteMask = 0;
14936 if (!(C & 0x000000ff))
14937 ZeroByteMask |= 0x000000ff;
14938 if (!(C & 0x0000ff00))
14939 ZeroByteMask |= 0x0000ff00;
14940 if (!(C & 0x00ff0000))
14941 ZeroByteMask |= 0x00ff0000;
14942 if (!(C & 0xff000000))
14943 ZeroByteMask |= 0xff000000;
14944 uint32_t NonZeroByteMask = ~ZeroByteMask; // 0xff for any non-zero byte
14945 if ((NonZeroByteMask & C) != NonZeroByteMask)
14946 return 0; // Partial bytes selected.
14947 return C;
14948}
14949
14950// Check if a node selects whole bytes from its operand 0 starting at a byte
14951// boundary while masking the rest. Returns select mask as in the v_perm_b32
14952// or -1 if not succeeded.
14953// Note byte select encoding:
14954// value 0-3 selects corresponding source byte;
14955// value 0xc selects zero;
14956// value 0xff selects 0xff.
14958 assert(V.getValueSizeInBits() == 32);
14959
14960 if (V.getNumOperands() != 2)
14961 return ~0;
14962
14963 ConstantSDNode *N1 = dyn_cast<ConstantSDNode>(V.getOperand(1));
14964 if (!N1)
14965 return ~0;
14966
14967 uint32_t C = N1->getZExtValue();
14968
14969 switch (V.getOpcode()) {
14970 default:
14971 break;
14972 case ISD::AND:
14973 if (uint32_t ConstMask = getConstantPermuteMask(C))
14974 return (0x03020100 & ConstMask) | (0x0c0c0c0c & ~ConstMask);
14975 break;
14976
14977 case ISD::OR:
14978 if (uint32_t ConstMask = getConstantPermuteMask(C))
14979 return (0x03020100 & ~ConstMask) | ConstMask;
14980 break;
14981
14982 case ISD::SHL:
14983 if (C % 8)
14984 return ~0;
14985
14986 return uint32_t((0x030201000c0c0c0cull << C) >> 32);
14987
14988 case ISD::SRL:
14989 if (C % 8)
14990 return ~0;
14991
14992 return uint32_t(0x0c0c0c0c03020100ull >> C);
14993 }
14994
14995 return ~0;
14996}
14997
14998SDValue SITargetLowering::performAndCombine(SDNode *N,
14999 DAGCombinerInfo &DCI) const {
15000 if (DCI.isBeforeLegalize())
15001 return SDValue();
15002
15003 SelectionDAG &DAG = DCI.DAG;
15004 EVT VT = N->getValueType(0);
15005 SDValue LHS = N->getOperand(0);
15006 SDValue RHS = N->getOperand(1);
15007
15008 const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(RHS);
15009 if (VT == MVT::i64 && CRHS) {
15010 if (SDValue Split =
15011 splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::AND, LHS, CRHS))
15012 return Split;
15013 }
15014
15015 if (CRHS && VT == MVT::i32) {
15016 // and (srl x, c), mask => shl (bfe x, nb + c, mask >> nb), nb
15017 // nb = number of trailing zeroes in mask
15018 // It can be optimized out using SDWA for GFX8+ in the SDWA peephole pass,
15019 // given that we are selecting 8 or 16 bit fields starting at byte boundary.
15020 uint64_t Mask = CRHS->getZExtValue();
15021 unsigned Bits = llvm::popcount(Mask);
15022 if (getSubtarget()->hasSDWA() && LHS->getOpcode() == ISD::SRL &&
15023 (Bits == 8 || Bits == 16) && isShiftedMask_64(Mask) && !(Mask & 1)) {
15024 if (auto *CShift = dyn_cast<ConstantSDNode>(LHS->getOperand(1))) {
15025 unsigned Shift = CShift->getZExtValue();
15026 unsigned NB = CRHS->getAPIntValue().countr_zero();
15027 unsigned Offset = NB + Shift;
15028 if ((Offset & (Bits - 1)) == 0) { // Starts at a byte or word boundary.
15029 SDLoc SL(N);
15030 SDValue BFE =
15031 DAG.getNode(AMDGPUISD::BFE_U32, SL, MVT::i32, LHS->getOperand(0),
15032 DAG.getConstant(Offset, SL, MVT::i32),
15033 DAG.getConstant(Bits, SL, MVT::i32));
15034 EVT NarrowVT = EVT::getIntegerVT(*DAG.getContext(), Bits);
15035 SDValue Ext = DAG.getNode(ISD::AssertZext, SL, VT, BFE,
15036 DAG.getValueType(NarrowVT));
15037 SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(LHS), VT, Ext,
15038 DAG.getConstant(NB, SDLoc(CRHS), MVT::i32));
15039 return Shl;
15040 }
15041 }
15042 }
15043
15044 // and (perm x, y, c1), c2 -> perm x, y, permute_mask(c1, c2)
15045 if (LHS.hasOneUse() && LHS.getOpcode() == AMDGPUISD::PERM &&
15046 isa<ConstantSDNode>(LHS.getOperand(2))) {
15047 uint32_t Sel = getConstantPermuteMask(Mask);
15048 if (!Sel)
15049 return SDValue();
15050
15051 // Select 0xc for all zero bytes
15052 Sel = (LHS.getConstantOperandVal(2) & Sel) | (~Sel & 0x0c0c0c0c);
15053 SDLoc DL(N);
15054 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, LHS.getOperand(0),
15055 LHS.getOperand(1), DAG.getConstant(Sel, DL, MVT::i32));
15056 }
15057 }
15058
15059 // (and (fcmp ord x, x), (fcmp une (fabs x), inf)) ->
15060 // fp_class x, ~(s_nan | q_nan | n_infinity | p_infinity)
15061 if (LHS.getOpcode() == ISD::SETCC && RHS.getOpcode() == ISD::SETCC) {
15062 ISD::CondCode LCC = cast<CondCodeSDNode>(LHS.getOperand(2))->get();
15063 ISD::CondCode RCC = cast<CondCodeSDNode>(RHS.getOperand(2))->get();
15064
15065 SDValue X = LHS.getOperand(0);
15066 SDValue Y = RHS.getOperand(0);
15067 if (Y.getOpcode() != ISD::FABS || Y.getOperand(0) != X ||
15068 !isTypeLegal(X.getValueType()))
15069 return SDValue();
15070
15071 if (LCC == ISD::SETO) {
15072 if (X != LHS.getOperand(1))
15073 return SDValue();
15074
15075 if (RCC == ISD::SETUNE) {
15076 const ConstantFPSDNode *C1 =
15077 dyn_cast<ConstantFPSDNode>(RHS.getOperand(1));
15078 if (!C1 || !C1->isInfinity() || C1->isNegative())
15079 return SDValue();
15080
15081 const uint32_t Mask = SIInstrFlags::N_NORMAL |
15085
15086 static_assert(
15089 0x3ff) == Mask,
15090 "mask not equal");
15091
15092 SDLoc DL(N);
15093 return DAG.getNode(AMDGPUISD::FP_CLASS, DL, MVT::i1, X,
15094 DAG.getConstant(Mask, DL, MVT::i32));
15095 }
15096 }
15097 }
15098
15099 if (RHS.getOpcode() == ISD::SETCC && LHS.getOpcode() == AMDGPUISD::FP_CLASS)
15100 std::swap(LHS, RHS);
15101
15102 if (LHS.getOpcode() == ISD::SETCC && RHS.getOpcode() == AMDGPUISD::FP_CLASS &&
15103 RHS.hasOneUse()) {
15104 ISD::CondCode LCC = cast<CondCodeSDNode>(LHS.getOperand(2))->get();
15105 // and (fcmp seto), (fp_class x, mask) -> fp_class x, mask & ~(p_nan |
15106 // n_nan) and (fcmp setuo), (fp_class x, mask) -> fp_class x, mask & (p_nan
15107 // | n_nan)
15108 const ConstantSDNode *Mask = dyn_cast<ConstantSDNode>(RHS.getOperand(1));
15109 if ((LCC == ISD::SETO || LCC == ISD::SETUO) && Mask &&
15110 (RHS.getOperand(0) == LHS.getOperand(0) &&
15111 LHS.getOperand(0) == LHS.getOperand(1))) {
15112 const unsigned OrdMask = SIInstrFlags::S_NAN | SIInstrFlags::Q_NAN;
15113 unsigned NewMask = LCC == ISD::SETO ? Mask->getZExtValue() & ~OrdMask
15114 : Mask->getZExtValue() & OrdMask;
15115
15116 SDLoc DL(N);
15117 return DAG.getNode(AMDGPUISD::FP_CLASS, DL, MVT::i1, RHS.getOperand(0),
15118 DAG.getConstant(NewMask, DL, MVT::i32));
15119 }
15120 }
15121
15122 if (VT == MVT::i32 && (RHS.getOpcode() == ISD::SIGN_EXTEND ||
15123 LHS.getOpcode() == ISD::SIGN_EXTEND)) {
15124 // and x, (sext cc from i1) => select cc, x, 0
15125 if (RHS.getOpcode() != ISD::SIGN_EXTEND)
15126 std::swap(LHS, RHS);
15127 if (isBoolSGPR(RHS.getOperand(0)))
15128 return DAG.getSelect(SDLoc(N), MVT::i32, RHS.getOperand(0), LHS,
15129 DAG.getConstant(0, SDLoc(N), MVT::i32));
15130 }
15131
15132 // and (op x, c1), (op y, c2) -> perm x, y, permute_mask(c1, c2)
15133 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
15134 if (VT == MVT::i32 && LHS.hasOneUse() && RHS.hasOneUse() &&
15135 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
15136 uint32_t LHSMask = getPermuteMask(LHS);
15137 uint32_t RHSMask = getPermuteMask(RHS);
15138 if (LHSMask != ~0u && RHSMask != ~0u) {
15139 // Canonicalize the expression in an attempt to have fewer unique masks
15140 // and therefore fewer registers used to hold the masks.
15141 if (LHSMask > RHSMask) {
15142 std::swap(LHSMask, RHSMask);
15143 std::swap(LHS, RHS);
15144 }
15145
15146 // Select 0xc for each lane used from source operand. Zero has 0xc mask
15147 // set, 0xff have 0xff in the mask, actual lanes are in the 0-3 range.
15148 uint32_t LHSUsedLanes = ~(LHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15149 uint32_t RHSUsedLanes = ~(RHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15150
15151 // Check of we need to combine values from two sources within a byte.
15152 if (!(LHSUsedLanes & RHSUsedLanes) &&
15153 // If we select high and lower word keep it for SDWA.
15154 // TODO: teach SDWA to work with v_perm_b32 and remove the check.
15155 !(LHSUsedLanes == 0x0c0c0000 && RHSUsedLanes == 0x00000c0c)) {
15156 // Each byte in each mask is either selector mask 0-3, or has higher
15157 // bits set in either of masks, which can be 0xff for 0xff or 0x0c for
15158 // zero. If 0x0c is in either mask it shall always be 0x0c. Otherwise
15159 // mask which is not 0xff wins. By anding both masks we have a correct
15160 // result except that 0x0c shall be corrected to give 0x0c only.
15161 uint32_t Mask = LHSMask & RHSMask;
15162 for (unsigned I = 0; I < 32; I += 8) {
15163 uint32_t ByteSel = 0xff << I;
15164 if ((LHSMask & ByteSel) == 0x0c || (RHSMask & ByteSel) == 0x0c)
15165 Mask &= (0x0c << I) & 0xffffffff;
15166 }
15167
15168 // Add 4 to each active LHS lane. It will not affect any existing 0xff
15169 // or 0x0c.
15170 uint32_t Sel = Mask | (LHSUsedLanes & 0x04040404);
15171 SDLoc DL(N);
15172
15173 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, LHS.getOperand(0),
15174 RHS.getOperand(0),
15175 DAG.getConstant(Sel, DL, MVT::i32));
15176 }
15177 }
15178 }
15179
15180 return SDValue();
15181}
15182
15183// A key component of v_perm is a mapping between byte position of the src
15184// operands, and the byte position of the dest. To provide such, we need: 1. the
15185// node that provides x byte of the dest of the OR, and 2. the byte of the node
15186// used to provide that x byte. calculateByteProvider finds which node provides
15187// a certain byte of the dest of the OR, and calculateSrcByte takes that node,
15188// and finds an ultimate src and byte position For example: The supported
15189// LoadCombine pattern for vector loads is as follows
15190// t1
15191// or
15192// / \
15193// t2 t3
15194// zext shl
15195// | | \
15196// t4 t5 16
15197// or anyext
15198// / \ |
15199// t6 t7 t8
15200// srl shl or
15201// / | / \ / \
15202// t9 t10 t11 t12 t13 t14
15203// trunc* 8 trunc* 8 and and
15204// | | / | | \
15205// t15 t16 t17 t18 t19 t20
15206// trunc* 255 srl -256
15207// | / \
15208// t15 t15 16
15209//
15210// *In this example, the truncs are from i32->i16
15211//
15212// calculateByteProvider would find t6, t7, t13, and t14 for bytes 0-3
15213// respectively. calculateSrcByte would find (given node) -> ultimate src &
15214// byteposition: t6 -> t15 & 1, t7 -> t16 & 0, t13 -> t15 & 0, t14 -> t15 & 3.
15215// After finding the mapping, we can combine the tree into vperm t15, t16,
15216// 0x05000407
15217
15218// Find the source and byte position from a node.
15219// \p DestByte is the byte position of the dest of the or that the src
15220// ultimately provides. \p SrcIndex is the byte of the src that maps to this
15221// dest of the or byte. \p Depth tracks how many recursive iterations we have
15222// performed.
15223static const std::optional<ByteProvider<SDValue>>
15224calculateSrcByte(const SDValue Op, uint64_t DestByte, uint64_t SrcIndex = 0,
15225 unsigned Depth = 0) {
15226 // We may need to recursively traverse a series of SRLs
15227 if (Depth >= 6)
15228 return std::nullopt;
15229
15230 if (Op.getValueSizeInBits() < 8)
15231 return std::nullopt;
15232
15233 if (Op.getValueType().isVector())
15234 return ByteProvider<SDValue>::getSrc(Op, DestByte, SrcIndex);
15235
15236 switch (Op->getOpcode()) {
15237 case ISD::TRUNCATE: {
15238 return calculateSrcByte(Op->getOperand(0), DestByte, SrcIndex, Depth + 1);
15239 }
15240
15241 case ISD::ANY_EXTEND:
15242 case ISD::SIGN_EXTEND:
15243 case ISD::ZERO_EXTEND:
15245 SDValue NarrowOp = Op->getOperand(0);
15246 auto NarrowVT = NarrowOp.getValueType();
15247 if (Op->getOpcode() == ISD::SIGN_EXTEND_INREG) {
15248 auto *VTSign = cast<VTSDNode>(Op->getOperand(1));
15249 NarrowVT = VTSign->getVT();
15250 }
15251 if (!NarrowVT.isByteSized())
15252 return std::nullopt;
15253 uint64_t NarrowByteWidth = NarrowVT.getStoreSize();
15254
15255 if (SrcIndex >= NarrowByteWidth)
15256 return std::nullopt;
15257 return calculateSrcByte(Op->getOperand(0), DestByte, SrcIndex, Depth + 1);
15258 }
15259
15260 case ISD::SRA:
15261 case ISD::SRL: {
15262 auto *ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
15263 if (!ShiftOp)
15264 return std::nullopt;
15265
15266 uint64_t BitShift = ShiftOp->getZExtValue();
15267
15268 if (BitShift % 8 != 0)
15269 return std::nullopt;
15270
15271 uint64_t NewSrcIndex = SrcIndex + BitShift / 8;
15272 if (NewSrcIndex >= Op.getScalarValueSizeInBits() / 8)
15273 return std::nullopt;
15274
15275 return calculateSrcByte(Op->getOperand(0), DestByte, NewSrcIndex,
15276 Depth + 1);
15277 }
15278
15279 default: {
15280 return ByteProvider<SDValue>::getSrc(Op, DestByte, SrcIndex);
15281 }
15282 }
15283 llvm_unreachable("fully handled switch");
15284}
15285
15286// For a byte position in the result of an Or, traverse the tree and find the
15287// node (and the byte of the node) which ultimately provides this {Or,
15288// BytePosition}. \p Op is the operand we are currently examining. \p Index is
15289// the byte position of the Op that corresponds with the originally requested
15290// byte of the Or \p Depth tracks how many recursive iterations we have
15291// performed. \p StartingIndex is the originally requested byte of the Or
15292static const std::optional<ByteProvider<SDValue>>
15293calculateByteProvider(const SDValue &Op, unsigned Index, unsigned Depth,
15294 unsigned StartingIndex = 0) {
15295 // Finding Src tree of RHS of or typically requires at least 1 additional
15296 // depth
15297 if (Depth > 6)
15298 return std::nullopt;
15299
15300 unsigned BitWidth = Op.getScalarValueSizeInBits();
15301 if (BitWidth % 8 != 0)
15302 return std::nullopt;
15303 if (Index > BitWidth / 8 - 1)
15304 return std::nullopt;
15305
15306 bool IsVec = Op.getValueType().isVector();
15307 switch (Op.getOpcode()) {
15308 case ISD::OR: {
15309 if (IsVec)
15310 return std::nullopt;
15311
15312 auto RHS = calculateByteProvider(Op.getOperand(1), Index, Depth + 1,
15313 StartingIndex);
15314 if (!RHS)
15315 return std::nullopt;
15316 auto LHS = calculateByteProvider(Op.getOperand(0), Index, Depth + 1,
15317 StartingIndex);
15318 if (!LHS)
15319 return std::nullopt;
15320 // A well formed Or will have two ByteProviders for each byte, one of which
15321 // is constant zero
15322 if (!LHS->isConstantZero() && !RHS->isConstantZero())
15323 return std::nullopt;
15324 if (!LHS || LHS->isConstantZero())
15325 return RHS;
15326 if (!RHS || RHS->isConstantZero())
15327 return LHS;
15328 return std::nullopt;
15329 }
15330
15331 case ISD::AND: {
15332 if (IsVec)
15333 return std::nullopt;
15334
15335 auto *BitMaskOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
15336 if (!BitMaskOp)
15337 return std::nullopt;
15338
15339 uint32_t BitMask = BitMaskOp->getZExtValue();
15340 // Bits we expect for our StartingIndex
15341 uint32_t IndexMask = 0xFF << (Index * 8);
15342
15343 if ((IndexMask & BitMask) != IndexMask) {
15344 // If the result of the and partially provides the byte, then it
15345 // is not well formatted
15346 if (IndexMask & BitMask)
15347 return std::nullopt;
15349 }
15350
15351 return calculateSrcByte(Op->getOperand(0), StartingIndex, Index);
15352 }
15353
15354 case ISD::FSHR: {
15355 if (IsVec)
15356 return std::nullopt;
15357
15358 // fshr(X,Y,Z): (X << (BW - (Z % BW))) | (Y >> (Z % BW))
15359 auto *ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(2));
15360 if (!ShiftOp || Op.getValueType().isVector())
15361 return std::nullopt;
15362
15363 uint64_t BitsProvided = Op.getValueSizeInBits();
15364 if (BitsProvided % 8 != 0)
15365 return std::nullopt;
15366
15367 uint64_t BitShift = ShiftOp->getAPIntValue().urem(BitsProvided);
15368 if (BitShift % 8)
15369 return std::nullopt;
15370
15371 uint64_t ConcatSizeInBytes = BitsProvided / 4;
15372 uint64_t ByteShift = BitShift / 8;
15373
15374 uint64_t NewIndex = (Index + ByteShift) % ConcatSizeInBytes;
15375 uint64_t BytesProvided = BitsProvided / 8;
15376 SDValue NextOp = Op.getOperand(NewIndex >= BytesProvided ? 0 : 1);
15377 NewIndex %= BytesProvided;
15378 return calculateByteProvider(NextOp, NewIndex, Depth + 1, StartingIndex);
15379 }
15380
15381 case ISD::SRA:
15382 case ISD::SRL: {
15383 if (IsVec)
15384 return std::nullopt;
15385
15386 auto *ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
15387 if (!ShiftOp)
15388 return std::nullopt;
15389
15390 uint64_t BitShift = ShiftOp->getZExtValue();
15391 if (BitShift % 8)
15392 return std::nullopt;
15393
15394 auto BitsProvided = Op.getScalarValueSizeInBits();
15395 if (BitsProvided % 8 != 0)
15396 return std::nullopt;
15397
15398 uint64_t BytesProvided = BitsProvided / 8;
15399 uint64_t ByteShift = BitShift / 8;
15400 if (Index + ByteShift < BytesProvided)
15401 return calculateSrcByte(Op->getOperand(0), StartingIndex,
15402 Index + ByteShift);
15403 // SRA's out-of-range bytes are sign bits, not constant zero.
15404 if (Op.getOpcode() == ISD::SRA)
15405 return std::nullopt;
15407 }
15408
15409 case ISD::SHL: {
15410 if (IsVec)
15411 return std::nullopt;
15412
15413 auto *ShiftOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
15414 if (!ShiftOp)
15415 return std::nullopt;
15416
15417 uint64_t BitShift = ShiftOp->getZExtValue();
15418 if (BitShift % 8 != 0)
15419 return std::nullopt;
15420 uint64_t ByteShift = BitShift / 8;
15421
15422 // If we are shifting by an amount greater than (or equal to)
15423 // the index we are trying to provide, then it provides 0s. If not,
15424 // then this bytes are not definitively 0s, and the corresponding byte
15425 // of interest is Index - ByteShift of the src
15426 return Index < ByteShift
15428 : calculateByteProvider(Op.getOperand(0), Index - ByteShift,
15429 Depth + 1, StartingIndex);
15430 }
15431 case ISD::ANY_EXTEND:
15432 case ISD::SIGN_EXTEND:
15433 case ISD::ZERO_EXTEND:
15435 case ISD::AssertZext:
15436 case ISD::AssertSext: {
15437 if (IsVec)
15438 return std::nullopt;
15439
15440 SDValue NarrowOp = Op->getOperand(0);
15441 unsigned NarrowBitWidth = NarrowOp.getValueSizeInBits();
15442 if (Op->getOpcode() == ISD::SIGN_EXTEND_INREG ||
15443 Op->getOpcode() == ISD::AssertZext ||
15444 Op->getOpcode() == ISD::AssertSext) {
15445 auto *VTSign = cast<VTSDNode>(Op->getOperand(1));
15446 NarrowBitWidth = VTSign->getVT().getSizeInBits();
15447 }
15448 if (NarrowBitWidth % 8 != 0)
15449 return std::nullopt;
15450 uint64_t NarrowByteWidth = NarrowBitWidth / 8;
15451
15452 if (Index >= NarrowByteWidth)
15453 return Op.getOpcode() == ISD::ZERO_EXTEND
15454 ? std::optional<ByteProvider<SDValue>>(
15456 : std::nullopt;
15457 return calculateByteProvider(NarrowOp, Index, Depth + 1, StartingIndex);
15458 }
15459
15460 case ISD::TRUNCATE: {
15461 if (IsVec)
15462 return std::nullopt;
15463
15464 uint64_t NarrowByteWidth = BitWidth / 8;
15465
15466 if (NarrowByteWidth >= Index) {
15467 return calculateByteProvider(Op.getOperand(0), Index, Depth + 1,
15468 StartingIndex);
15469 }
15470
15471 return std::nullopt;
15472 }
15473
15474 case ISD::CopyFromReg: {
15475 if (BitWidth / 8 > Index)
15476 return calculateSrcByte(Op, StartingIndex, Index);
15477
15478 return std::nullopt;
15479 }
15480
15481 case ISD::LOAD: {
15482 auto *L = cast<LoadSDNode>(Op.getNode());
15483
15484 unsigned NarrowBitWidth = L->getMemoryVT().getSizeInBits();
15485 if (NarrowBitWidth % 8 != 0)
15486 return std::nullopt;
15487 uint64_t NarrowByteWidth = NarrowBitWidth / 8;
15488
15489 // If the width of the load does not reach byte we are trying to provide for
15490 // and it is not a ZEXTLOAD, then the load does not provide for the byte in
15491 // question
15492 if (Index >= NarrowByteWidth) {
15493 return L->getExtensionType() == ISD::ZEXTLOAD
15494 ? std::optional<ByteProvider<SDValue>>(
15496 : std::nullopt;
15497 }
15498
15499 if (NarrowByteWidth > Index) {
15500 return calculateSrcByte(Op, StartingIndex, Index);
15501 }
15502
15503 return std::nullopt;
15504 }
15505
15506 case ISD::BSWAP: {
15507 if (IsVec)
15508 return std::nullopt;
15509
15510 return calculateByteProvider(Op->getOperand(0), BitWidth / 8 - Index - 1,
15511 Depth + 1, StartingIndex);
15512 }
15513
15515 auto *IdxOp = dyn_cast<ConstantSDNode>(Op->getOperand(1));
15516 if (!IdxOp)
15517 return std::nullopt;
15518 auto VecIdx = IdxOp->getZExtValue();
15519 auto ScalarSize = Op.getScalarValueSizeInBits();
15520 if (ScalarSize < 32)
15521 Index = ScalarSize == 8 ? VecIdx : VecIdx * 2 + Index;
15522 return calculateSrcByte(ScalarSize >= 32 ? Op : Op.getOperand(0),
15523 StartingIndex, Index);
15524 }
15525
15526 case AMDGPUISD::PERM: {
15527 if (IsVec)
15528 return std::nullopt;
15529
15530 auto *PermMask = dyn_cast<ConstantSDNode>(Op->getOperand(2));
15531 if (!PermMask)
15532 return std::nullopt;
15533
15534 auto IdxMask =
15535 (PermMask->getZExtValue() & (0xFF << (Index * 8))) >> (Index * 8);
15536 if (IdxMask > 0x07 && IdxMask != 0x0c)
15537 return std::nullopt;
15538
15539 auto NextOp = Op.getOperand(IdxMask > 0x03 ? 0 : 1);
15540 auto NextIndex = IdxMask > 0x03 ? IdxMask % 4 : IdxMask;
15541
15542 return IdxMask != 0x0c ? calculateSrcByte(NextOp, StartingIndex, NextIndex)
15545 }
15546
15547 default: {
15548 return std::nullopt;
15549 }
15550 }
15551
15552 llvm_unreachable("fully handled switch");
15553}
15554
15555// Returns true if the Operand is a scalar and is 16 bits
15556static bool isExtendedFrom16Bits(SDValue &Operand) {
15557
15558 switch (Operand.getOpcode()) {
15559 case ISD::ANY_EXTEND:
15560 case ISD::SIGN_EXTEND:
15561 case ISD::ZERO_EXTEND: {
15562 auto OpVT = Operand.getOperand(0).getValueType();
15563 return !OpVT.isVector() && OpVT.getSizeInBits() == 16;
15564 }
15565 case ISD::LOAD: {
15566 LoadSDNode *L = cast<LoadSDNode>(Operand.getNode());
15567 auto ExtType = cast<LoadSDNode>(L)->getExtensionType();
15568 if (ExtType == ISD::ZEXTLOAD || ExtType == ISD::SEXTLOAD ||
15569 ExtType == ISD::EXTLOAD) {
15570 auto MemVT = L->getMemoryVT();
15571 return !MemVT.isVector() && MemVT.getSizeInBits() == 16;
15572 }
15573 return L->getMemoryVT().getSizeInBits() == 16;
15574 }
15575 default:
15576 return false;
15577 }
15578}
15579
15580// Returns true if the mask matches consecutive bytes, and the first byte
15581// begins at a power of 2 byte offset from 0th byte
15582static bool addresses16Bits(int Mask) {
15583 int Low8 = Mask & 0xff;
15584 int Hi8 = (Mask & 0xff00) >> 8;
15585
15586 assert(Low8 < 8 && Hi8 < 8);
15587 // Are the bytes contiguous in the order of increasing addresses.
15588 bool IsConsecutive = (Hi8 - Low8 == 1);
15589 // Is the first byte at location that is aligned for 16 bit instructions.
15590 // A counter example is taking 2 consecutive bytes starting at the 8th bit.
15591 // In this case, we still need code to extract the 16 bit operand, so it
15592 // is better to use i8 v_perm
15593 bool Is16Aligned = !(Low8 % 2);
15594
15595 return IsConsecutive && Is16Aligned;
15596}
15597
15598// Do not lower into v_perm if the operands are actually 16 bit
15599// and the selected bits (based on PermMask) correspond with two
15600// easily addressable 16 bit operands.
15602 SDValue &OtherOp) {
15603 int Low16 = PermMask & 0xffff;
15604 int Hi16 = (PermMask & 0xffff0000) >> 16;
15605
15606 auto TempOp = peekThroughBitcasts(Op);
15607 auto TempOtherOp = peekThroughBitcasts(OtherOp);
15608
15609 auto OpIs16Bit =
15610 TempOp.getValueSizeInBits() == 16 || isExtendedFrom16Bits(TempOp);
15611 if (!OpIs16Bit)
15612 return true;
15613
15614 auto OtherOpIs16Bit = TempOtherOp.getValueSizeInBits() == 16 ||
15615 isExtendedFrom16Bits(TempOtherOp);
15616 if (!OtherOpIs16Bit)
15617 return true;
15618
15619 // Do we cleanly address both
15620 return !addresses16Bits(Low16) || !addresses16Bits(Hi16);
15621}
15622
15624 unsigned DWordOffset) {
15625 SDValue Ret;
15626
15627 auto TypeSize = Src.getValueSizeInBits().getFixedValue();
15628 // ByteProvider must be at least 8 bits
15629 assert(Src.getValueSizeInBits().isKnownMultipleOf(8));
15630
15631 if (TypeSize <= 32)
15632 return DAG.getBitcastedAnyExtOrTrunc(Src, SL, MVT::i32);
15633
15634 if (Src.getValueType().isVector()) {
15635 auto ScalarTySize = Src.getScalarValueSizeInBits();
15636 auto ScalarTy = Src.getValueType().getScalarType();
15637 if (ScalarTySize == 32) {
15638 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Src,
15639 DAG.getConstant(DWordOffset, SL, MVT::i32));
15640 }
15641 if (ScalarTySize > 32) {
15642 Ret = DAG.getNode(
15643 ISD::EXTRACT_VECTOR_ELT, SL, ScalarTy, Src,
15644 DAG.getConstant(DWordOffset / (ScalarTySize / 32), SL, MVT::i32));
15645 auto ShiftVal = 32 * (DWordOffset % (ScalarTySize / 32));
15646 if (ShiftVal)
15647 Ret = DAG.getNode(ISD::SRL, SL, Ret.getValueType(), Ret,
15648 DAG.getConstant(ShiftVal, SL, MVT::i32));
15649 return DAG.getBitcastedAnyExtOrTrunc(Ret, SL, MVT::i32);
15650 }
15651
15652 assert(ScalarTySize < 32);
15653 if (TypeSize % 32 == 0) {
15654 assert(DWordOffset < TypeSize / 32);
15655 SDValue Cast = DAG.getBitcast(
15656 EVT::getVectorVT(*DAG.getContext(), MVT::i32, TypeSize / 32), Src);
15657 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Cast,
15658 DAG.getConstant(DWordOffset, SL, MVT::i32));
15659 }
15660
15661 auto NumElements = TypeSize / ScalarTySize;
15662 auto Trunc32Elements = (ScalarTySize * NumElements) / 32;
15663 auto NormalizedTrunc = Trunc32Elements * 32 / ScalarTySize;
15664 auto NumElementsIn32 = 32 / ScalarTySize;
15665 auto NumAvailElements = DWordOffset < Trunc32Elements
15666 ? NumElementsIn32
15667 : NumElements - NormalizedTrunc;
15668
15670 DAG.ExtractVectorElements(Src, VecSrcs, DWordOffset * NumElementsIn32,
15671 NumAvailElements);
15672
15673 Ret = DAG.getBuildVector(
15674 MVT::getVectorVT(MVT::getIntegerVT(ScalarTySize), NumAvailElements), SL,
15675 VecSrcs);
15676 return Ret = DAG.getBitcastedAnyExtOrTrunc(Ret, SL, MVT::i32);
15677 }
15678
15679 /// Scalar Type
15680 auto ShiftVal = 32 * DWordOffset;
15681 Ret = DAG.getNode(ISD::SRL, SL, Src.getValueType(), Src,
15682 DAG.getConstant(ShiftVal, SL, MVT::i32));
15683 return DAG.getBitcastedAnyExtOrTrunc(Ret, SL, MVT::i32);
15684}
15685
15687 SelectionDAG &DAG = DCI.DAG;
15688 [[maybe_unused]] EVT VT = N->getValueType(0);
15690
15691 // VT is known to be MVT::i32, so we need to provide 4 bytes.
15692 assert(VT == MVT::i32);
15693 for (int i = 0; i < 4; i++) {
15694 // Find the ByteProvider that provides the ith byte of the result of OR
15695 std::optional<ByteProvider<SDValue>> P =
15696 calculateByteProvider(SDValue(N, 0), i, 0, /*StartingIndex = */ i);
15697 // TODO support constantZero
15698 if (!P || P->isConstantZero())
15699 return SDValue();
15700
15701 PermNodes.push_back(*P);
15702 }
15703 if (PermNodes.size() != 4)
15704 return SDValue();
15705
15706 std::pair<unsigned, unsigned> FirstSrc(0, PermNodes[0].SrcOffset / 4);
15707 std::optional<std::pair<unsigned, unsigned>> SecondSrc;
15708 uint64_t PermMask = 0x00000000;
15709 for (size_t i = 0; i < PermNodes.size(); i++) {
15710 auto PermOp = PermNodes[i];
15711 // Since the mask is applied to Src1:Src2, Src1 bytes must be offset
15712 // by sizeof(Src2) = 4
15713 int SrcByteAdjust = 4;
15714
15715 // If the Src uses a byte from a different DWORD, then it corresponds
15716 // with a difference source
15717 if (!PermOp.hasSameSrc(PermNodes[FirstSrc.first]) ||
15718 ((PermOp.SrcOffset / 4) != FirstSrc.second)) {
15719 if (SecondSrc)
15720 if (!PermOp.hasSameSrc(PermNodes[SecondSrc->first]) ||
15721 ((PermOp.SrcOffset / 4) != SecondSrc->second))
15722 return SDValue();
15723
15724 // Set the index of the second distinct Src node
15725 SecondSrc = {i, PermNodes[i].SrcOffset / 4};
15726 assert(!(PermNodes[SecondSrc->first].Src->getValueSizeInBits() % 8));
15727 SrcByteAdjust = 0;
15728 }
15729 assert((PermOp.SrcOffset % 4) + SrcByteAdjust < 8);
15731 PermMask |= ((PermOp.SrcOffset % 4) + SrcByteAdjust) << (i * 8);
15732 }
15733 SDLoc DL(N);
15734 SDValue Op = *PermNodes[FirstSrc.first].Src;
15735 Op = getDWordFromOffset(DAG, DL, Op, FirstSrc.second);
15736 assert(Op.getValueSizeInBits() == 32);
15737
15738 // Check that we are not just extracting the bytes in order from an op
15739 if (!SecondSrc) {
15740 int Low16 = PermMask & 0xffff;
15741 int Hi16 = (PermMask & 0xffff0000) >> 16;
15742
15743 bool WellFormedLow = (Low16 == 0x0504) || (Low16 == 0x0100);
15744 bool WellFormedHi = (Hi16 == 0x0706) || (Hi16 == 0x0302);
15745
15746 // The perm op would really just produce Op. So combine into Op
15747 if (WellFormedLow && WellFormedHi)
15748 return DAG.getBitcast(MVT::getIntegerVT(32), Op);
15749 }
15750
15751 SDValue OtherOp = SecondSrc ? *PermNodes[SecondSrc->first].Src : Op;
15752
15753 if (SecondSrc) {
15754 OtherOp = getDWordFromOffset(DAG, DL, OtherOp, SecondSrc->second);
15755 assert(OtherOp.getValueSizeInBits() == 32);
15756 }
15757
15758 // Check that we haven't just recreated the same FSHR node.
15759 if (N->getOpcode() == ISD::FSHR &&
15760 (N->getOperand(0) == Op || N->getOperand(0) == OtherOp) &&
15761 (N->getOperand(1) == Op || N->getOperand(1) == OtherOp))
15762 return SDValue();
15763
15764 if (hasNon16BitAccesses(PermMask, Op, OtherOp)) {
15765
15766 assert(Op.getValueType().isByteSized() &&
15767 OtherOp.getValueType().isByteSized());
15768
15769 // If the ultimate src is less than 32 bits, then we will only be
15770 // using bytes 0: Op.getValueSizeInBytes() - 1 in the or.
15771 // CalculateByteProvider would not have returned Op as source if we
15772 // used a byte that is outside its ValueType. Thus, we are free to
15773 // ANY_EXTEND as the extended bits are dont-cares.
15774 Op = DAG.getBitcastedAnyExtOrTrunc(Op, DL, MVT::i32);
15775 OtherOp = DAG.getBitcastedAnyExtOrTrunc(OtherOp, DL, MVT::i32);
15776
15777 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, Op, OtherOp,
15778 DAG.getConstant(PermMask, DL, MVT::i32));
15779 }
15780 return SDValue();
15781}
15782
15783SDValue SITargetLowering::performOrCombine(SDNode *N,
15784 DAGCombinerInfo &DCI) const {
15785 SelectionDAG &DAG = DCI.DAG;
15786 SDValue LHS = N->getOperand(0);
15787 SDValue RHS = N->getOperand(1);
15788
15789 EVT VT = N->getValueType(0);
15790 if (VT == MVT::i1) {
15791 // or (fp_class x, c1), (fp_class x, c2) -> fp_class x, (c1 | c2)
15792 if (LHS.getOpcode() == AMDGPUISD::FP_CLASS &&
15793 RHS.getOpcode() == AMDGPUISD::FP_CLASS) {
15794 SDValue Src = LHS.getOperand(0);
15795 if (Src != RHS.getOperand(0))
15796 return SDValue();
15797
15798 const ConstantSDNode *CLHS = dyn_cast<ConstantSDNode>(LHS.getOperand(1));
15799 const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(RHS.getOperand(1));
15800 if (!CLHS || !CRHS)
15801 return SDValue();
15802
15803 // Only 10 bits are used.
15804 static const uint32_t MaxMask = 0x3ff;
15805
15806 uint32_t NewMask =
15807 (CLHS->getZExtValue() | CRHS->getZExtValue()) & MaxMask;
15808 SDLoc DL(N);
15809 return DAG.getNode(AMDGPUISD::FP_CLASS, DL, MVT::i1, Src,
15810 DAG.getConstant(NewMask, DL, MVT::i32));
15811 }
15812
15813 return SDValue();
15814 }
15815
15816 // or (perm x, y, c1), c2 -> perm x, y, permute_mask(c1, c2)
15818 LHS.getOpcode() == AMDGPUISD::PERM &&
15819 isa<ConstantSDNode>(LHS.getOperand(2))) {
15820 uint32_t Sel = getConstantPermuteMask(N->getConstantOperandVal(1));
15821 if (!Sel)
15822 return SDValue();
15823
15824 Sel |= LHS.getConstantOperandVal(2);
15825 SDLoc DL(N);
15826 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, LHS.getOperand(0),
15827 LHS.getOperand(1), DAG.getConstant(Sel, DL, MVT::i32));
15828 }
15829
15830 // or (op x, c1), (op y, c2) -> perm x, y, permute_mask(c1, c2)
15831 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
15832 if (VT == MVT::i32 && LHS.hasOneUse() && RHS.hasOneUse() &&
15833 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
15834
15835 // If all the uses of an or need to extract the individual elements, do not
15836 // attempt to lower into v_perm
15837 auto usesCombinedOperand = [](SDNode *OrUse) {
15838 // If we have any non-vectorized use, then it is a candidate for v_perm
15839 if (OrUse->getOpcode() != ISD::BITCAST ||
15840 !OrUse->getValueType(0).isVector())
15841 return true;
15842
15843 // If we have any non-vectorized use, then it is a candidate for v_perm
15844 for (auto *VUser : OrUse->users()) {
15845 if (!VUser->getValueType(0).isVector())
15846 return true;
15847
15848 // If the use of a vector is a store, then combining via a v_perm
15849 // is beneficial.
15850 // TODO -- whitelist more uses
15851 for (auto VectorwiseOp : {ISD::STORE, ISD::CopyToReg, ISD::CopyFromReg})
15852 if (VUser->getOpcode() == VectorwiseOp)
15853 return true;
15854 }
15855 return false;
15856 };
15857
15858 if (!any_of(N->users(), usesCombinedOperand))
15859 return SDValue();
15860
15861 uint32_t LHSMask = getPermuteMask(LHS);
15862 uint32_t RHSMask = getPermuteMask(RHS);
15863
15864 if (LHSMask != ~0u && RHSMask != ~0u) {
15865 // Canonicalize the expression in an attempt to have fewer unique masks
15866 // and therefore fewer registers used to hold the masks.
15867 if (LHSMask > RHSMask) {
15868 std::swap(LHSMask, RHSMask);
15869 std::swap(LHS, RHS);
15870 }
15871
15872 // Select 0xc for each lane used from source operand. Zero has 0xc mask
15873 // set, 0xff have 0xff in the mask, actual lanes are in the 0-3 range.
15874 uint32_t LHSUsedLanes = ~(LHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15875 uint32_t RHSUsedLanes = ~(RHSMask & 0x0c0c0c0c) & 0x0c0c0c0c;
15876
15877 // Check of we need to combine values from two sources within a byte.
15878 if (!(LHSUsedLanes & RHSUsedLanes) &&
15879 // If we select high and lower word keep it for SDWA.
15880 // TODO: teach SDWA to work with v_perm_b32 and remove the check.
15881 !(LHSUsedLanes == 0x0c0c0000 && RHSUsedLanes == 0x00000c0c)) {
15882 // Kill zero bytes selected by other mask. Zero value is 0xc.
15883 LHSMask &= ~RHSUsedLanes;
15884 RHSMask &= ~LHSUsedLanes;
15885 // Add 4 to each active LHS lane
15886 LHSMask |= LHSUsedLanes & 0x04040404;
15887 // Combine masks
15888 uint32_t Sel = LHSMask | RHSMask;
15889 SDLoc DL(N);
15890
15891 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, LHS.getOperand(0),
15892 RHS.getOperand(0),
15893 DAG.getConstant(Sel, DL, MVT::i32));
15894 }
15895 }
15896 if (LHSMask == ~0u || RHSMask == ~0u) {
15897 if (SDValue Perm = matchPERM(N, DCI))
15898 return Perm;
15899 }
15900 }
15901
15902 // Detect identity v2i32 OR and replace with identity source node.
15903 // Specifically an Or that has operands constructed from the same source node
15904 // via extract_vector_elt and build_vector. I.E.
15905 // v2i32 or(
15906 // v2i32 build_vector(
15907 // i32 extract_elt(%IdentitySrc, 0),
15908 // i32 0
15909 // ),
15910 // v2i32 build_vector(
15911 // i32 0,
15912 // i32 extract_elt(%IdentitySrc, 1)
15913 // ) )
15914 // =>
15915 // v2i32 %IdentitySrc
15916
15917 if (VT == MVT::v2i32 && LHS->getOpcode() == ISD::BUILD_VECTOR &&
15918 RHS->getOpcode() == ISD::BUILD_VECTOR) {
15919
15920 ConstantSDNode *LC = dyn_cast<ConstantSDNode>(LHS->getOperand(1));
15921 ConstantSDNode *RC = dyn_cast<ConstantSDNode>(RHS->getOperand(0));
15922
15923 // Test for and normalise build vectors.
15924 if (LC && RC && LC->getZExtValue() == 0 && RC->getZExtValue() == 0) {
15925
15926 // Get the extract_vector_element operands.
15927 SDValue LEVE = LHS->getOperand(0);
15928 SDValue REVE = RHS->getOperand(1);
15929
15930 if (LEVE->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
15932 // Check that different elements from the same vector are
15933 // extracted.
15934 if (LEVE->getOperand(0) == REVE->getOperand(0) &&
15935 LEVE->getOperand(1) != REVE->getOperand(1)) {
15936 SDValue IdentitySrc = LEVE.getOperand(0);
15937 return IdentitySrc;
15938 }
15939 }
15940 }
15941 }
15942
15943 if (VT != MVT::i64 || DCI.isBeforeLegalizeOps())
15944 return SDValue();
15945
15946 // TODO: This could be a generic combine with a predicate for extracting the
15947 // high half of an integer being free.
15948
15949 // (or i64:x, (zero_extend i32:y)) ->
15950 // i64 (bitcast (v2i32 build_vector (or i32:y, lo_32(x)), hi_32(x)))
15951 if (LHS.getOpcode() == ISD::ZERO_EXTEND &&
15952 RHS.getOpcode() != ISD::ZERO_EXTEND)
15953 std::swap(LHS, RHS);
15954
15955 if (RHS.getOpcode() == ISD::ZERO_EXTEND) {
15956 SDValue ExtSrc = RHS.getOperand(0);
15957 EVT SrcVT = ExtSrc.getValueType();
15958 if (SrcVT == MVT::i32) {
15959 SDLoc SL(N);
15960 auto [LowLHS, HiBits] = split64BitValue(LHS, DAG);
15961 SDValue LowOr = DAG.getNode(ISD::OR, SL, MVT::i32, LowLHS, ExtSrc);
15962
15963 DCI.AddToWorklist(LowOr.getNode());
15964 DCI.AddToWorklist(HiBits.getNode());
15965
15966 SDValue Vec =
15967 DAG.getNode(ISD::BUILD_VECTOR, SL, MVT::v2i32, LowOr, HiBits);
15968 return DAG.getNode(ISD::BITCAST, SL, MVT::i64, Vec);
15969 }
15970 }
15971
15972 const ConstantSDNode *CRHS = dyn_cast<ConstantSDNode>(N->getOperand(1));
15973 if (CRHS) {
15974 if (SDValue Split = splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::OR,
15975 N->getOperand(0), CRHS))
15976 return Split;
15977 }
15978
15979 return SDValue();
15980}
15981
15982SDValue SITargetLowering::performXorCombine(SDNode *N,
15983 DAGCombinerInfo &DCI) const {
15984 if (SDValue RV = reassociateScalarOps(N, DCI.DAG))
15985 return RV;
15986
15987 SDValue LHS = N->getOperand(0);
15988 SDValue RHS = N->getOperand(1);
15989
15990 const ConstantSDNode *CRHS = isConstOrConstSplat(RHS);
15991 SelectionDAG &DAG = DCI.DAG;
15992
15993 EVT VT = N->getValueType(0);
15994 if (CRHS && VT == MVT::i64) {
15995 if (SDValue Split =
15996 splitBinaryBitConstantOp(DCI, SDLoc(N), ISD::XOR, LHS, CRHS))
15997 return Split;
15998 }
15999
16000 // v2i32 (xor (vselect cc, x, y), K) ->
16001 // (v2i32 svelect cc, (xor x, K), (xor y, K)) This enables the xor to be
16002 // replaced with source modifiers when the select is lowered to CNDMASK.
16003 unsigned Opc = LHS.getOpcode();
16004 if (((Opc == ISD::VSELECT && VT == MVT::v2i32) ||
16005 (Opc == ISD::SELECT && VT == MVT::i64)) &&
16006 CRHS && CRHS->getAPIntValue().isSignMask()) {
16007 SDValue CC = LHS->getOperand(0);
16008 SDValue TRUE = LHS->getOperand(1);
16009 SDValue FALSE = LHS->getOperand(2);
16010 SDValue XTrue = DAG.getNode(ISD::XOR, SDLoc(N), VT, TRUE, RHS);
16011 SDValue XFalse = DAG.getNode(ISD::XOR, SDLoc(N), VT, FALSE, RHS);
16012 SDValue XSelect =
16013 DAG.getNode(ISD::VSELECT, SDLoc(N), VT, CC, XTrue, XFalse);
16014 return XSelect;
16015 }
16016
16017 // Make sure to apply the 64-bit constant splitting fold before trying to fold
16018 // fneg-like xors into 64-bit select.
16019 if (LHS.getOpcode() == ISD::SELECT && VT == MVT::i32) {
16020 // This looks like an fneg, try to fold as a source modifier.
16021 if (CRHS && CRHS->getAPIntValue().isSignMask() &&
16023 // xor (select c, a, b), 0x80000000 ->
16024 // bitcast (select c, (fneg (bitcast a)), (fneg (bitcast b)))
16025 SDLoc DL(N);
16026 SDValue CastLHS =
16027 DAG.getNode(ISD::BITCAST, DL, MVT::f32, LHS->getOperand(1));
16028 SDValue CastRHS =
16029 DAG.getNode(ISD::BITCAST, DL, MVT::f32, LHS->getOperand(2));
16030 SDValue FNegLHS = DAG.getNode(ISD::FNEG, DL, MVT::f32, CastLHS);
16031 SDValue FNegRHS = DAG.getNode(ISD::FNEG, DL, MVT::f32, CastRHS);
16032 SDValue NewSelect = DAG.getNode(ISD::SELECT, DL, MVT::f32,
16033 LHS->getOperand(0), FNegLHS, FNegRHS);
16034 return DAG.getNode(ISD::BITCAST, DL, VT, NewSelect);
16035 }
16036 }
16037
16038 return SDValue();
16039}
16040
16041SDValue
16042SITargetLowering::performZeroOrAnyExtendCombine(SDNode *N,
16043 DAGCombinerInfo &DCI) const {
16044 if (!Subtarget->has16BitInsts() ||
16045 DCI.getDAGCombineLevel() < AfterLegalizeTypes)
16046 return SDValue();
16047
16048 EVT VT = N->getValueType(0);
16049 if (VT != MVT::i32)
16050 return SDValue();
16051
16052 SDValue Src = N->getOperand(0);
16053 if (Src.getValueType() != MVT::i16)
16054 return SDValue();
16055
16056 if (!Src->hasOneUse())
16057 return SDValue();
16058
16059 // TODO: We bail out below if SrcOffset is not in the first dword (>= 4). It's
16060 // possible we're missing out on some combine opportunities, but we'd need to
16061 // weigh the cost of extracting the byte from the upper dwords.
16062
16063 std::optional<ByteProvider<SDValue>> BP0 =
16064 calculateByteProvider(SDValue(N, 0), 0, 0, 0);
16065 if (!BP0 || BP0->SrcOffset >= 4 || !BP0->Src)
16066 return SDValue();
16067 SDValue V0 = *BP0->Src;
16068
16069 std::optional<ByteProvider<SDValue>> BP1 =
16070 calculateByteProvider(SDValue(N, 0), 1, 0, 1);
16071 if (!BP1 || BP1->SrcOffset >= 4 || !BP1->Src)
16072 return SDValue();
16073
16074 SDValue V1 = *BP1->Src;
16075
16076 if (V0 == V1)
16077 return SDValue();
16078
16079 SelectionDAG &DAG = DCI.DAG;
16080 SDLoc DL(N);
16081 uint32_t PermMask = 0x0c0c0c0c;
16082 if (V0) {
16083 V0 = DAG.getBitcastedAnyExtOrTrunc(V0, DL, MVT::i32);
16084 PermMask = (PermMask & ~0xFF) | (BP0->SrcOffset + 4);
16085 }
16086
16087 if (V1) {
16088 V1 = DAG.getBitcastedAnyExtOrTrunc(V1, DL, MVT::i32);
16089 PermMask = (PermMask & ~(0xFF << 8)) | (BP1->SrcOffset << 8);
16090 }
16091
16092 return DAG.getNode(AMDGPUISD::PERM, DL, MVT::i32, V0, V1,
16093 DAG.getConstant(PermMask, DL, MVT::i32));
16094}
16095
16096SDValue
16097SITargetLowering::performSignExtendInRegCombine(SDNode *N,
16098 DAGCombinerInfo &DCI) const {
16099 SDValue Src = N->getOperand(0);
16100 auto *VTSign = cast<VTSDNode>(N->getOperand(1));
16101
16102 // Combine s_buffer_load_u8 or s_buffer_load_u16 with sext and replace them
16103 // with s_buffer_load_i8 and s_buffer_load_i16 respectively.
16104 if (((Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_UBYTE &&
16105 VTSign->getVT() == MVT::i8) ||
16106 (Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_USHORT &&
16107 VTSign->getVT() == MVT::i16))) {
16108 assert(Subtarget->hasScalarSubwordLoads() &&
16109 "s_buffer_load_{u8, i8} are supported "
16110 "in GFX12 (or newer) architectures.");
16111 unsigned Opc = (Src.getOpcode() == AMDGPUISD::SBUFFER_LOAD_UBYTE)
16112 ? AMDGPUISD::SBUFFER_LOAD_BYTE
16113 : AMDGPUISD::SBUFFER_LOAD_SHORT;
16114 SDLoc DL(N);
16115 SDVTList ResList =
16116 DCI.DAG.getVTList(MVT::i32, Src.getOperand(0).getValueType());
16117 SDValue Ops[] = {
16118 Src.getOperand(0), // Chain
16119 Src.getOperand(1), // source register
16120 Src.getOperand(2), // offset
16121 Src.getOperand(3) // cachePolicy
16122 };
16123 auto *M = cast<MemSDNode>(Src);
16124 SDValue BufferLoad = DCI.DAG.getMemIntrinsicNode(
16125 Opc, DL, ResList, Ops, M->getMemoryVT(), M->getMemOperand());
16126 return DCI.DAG.getMergeValues({BufferLoad, BufferLoad.getValue(1)}, DL);
16127 }
16128 if (((Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_UBYTE &&
16129 VTSign->getVT() == MVT::i8) ||
16130 (Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_USHORT &&
16131 VTSign->getVT() == MVT::i16)) &&
16132 Src.hasOneUse()) {
16133 auto *M = cast<MemSDNode>(Src);
16134 SDValue Ops[] = {Src.getOperand(0), // Chain
16135 Src.getOperand(1), // rsrc
16136 Src.getOperand(2), // vindex
16137 Src.getOperand(3), // voffset
16138 Src.getOperand(4), // soffset
16139 Src.getOperand(5), // offset
16140 Src.getOperand(6), Src.getOperand(7)};
16141 // replace with BUFFER_LOAD_BYTE/SHORT
16142 SDVTList ResList =
16143 DCI.DAG.getVTList(MVT::i32, Src.getOperand(0).getValueType());
16144 unsigned Opc = (Src.getOpcode() == AMDGPUISD::BUFFER_LOAD_UBYTE)
16145 ? AMDGPUISD::BUFFER_LOAD_BYTE
16146 : AMDGPUISD::BUFFER_LOAD_SHORT;
16147 SDValue BufferLoadSignExt = DCI.DAG.getMemIntrinsicNode(
16148 Opc, SDLoc(N), ResList, Ops, M->getMemoryVT(), M->getMemOperand());
16149 return DCI.DAG.getMergeValues(
16150 {BufferLoadSignExt, BufferLoadSignExt.getValue(1)}, SDLoc(N));
16151 }
16152 return SDValue();
16153}
16154
16155SDValue SITargetLowering::performClassCombine(SDNode *N,
16156 DAGCombinerInfo &DCI) const {
16157 SelectionDAG &DAG = DCI.DAG;
16158 SDValue Mask = N->getOperand(1);
16159
16160 // fp_class x, 0 -> false
16161 if (isNullConstant(Mask))
16162 return DAG.getConstant(0, SDLoc(N), MVT::i1);
16163
16164 if (N->getOperand(0).isUndef())
16165 return DAG.getUNDEF(MVT::i1);
16166
16167 return SDValue();
16168}
16169
16170SDValue SITargetLowering::performRcpCombine(SDNode *N,
16171 DAGCombinerInfo &DCI) const {
16172 EVT VT = N->getValueType(0);
16173 SDValue N0 = N->getOperand(0);
16174
16175 if (N0.isUndef()) {
16176 return DCI.DAG.getConstantFP(APFloat::getQNaN(VT.getFltSemantics()),
16177 SDLoc(N), VT);
16178 }
16179
16180 // TODO: Could handle f32 + amdgcn.sqrt but probably never reaches here.
16181 if ((VT == MVT::f16 && N0.getOpcode() == ISD::FSQRT) &&
16182 N->getFlags().hasAllowContract() && N0->getFlags().hasAllowContract()) {
16183 return DCI.DAG.getNode(AMDGPUISD::RSQ, SDLoc(N), VT, N0.getOperand(0),
16184 N->getFlags());
16185 }
16186
16188}
16189
16191 SDNodeFlags UserFlags,
16192 unsigned MaxDepth) const {
16193 EVT VT = Op.getValueType();
16194 assert(VT.isFloatingPoint() &&
16195 "expected a floating-point value to query canonicality of");
16196 return isCanonicalized(DAG, Op, VT.getScalarType(), UserFlags, MaxDepth);
16197}
16198
16200 EVT QueryVT, SDNodeFlags UserFlags,
16201 unsigned MaxDepth) const {
16202 assert(QueryVT.isFloatingPoint() && !QueryVT.isVector() &&
16203 "QueryVT must be a floating-point scalar type");
16204 EVT VT = Op.getValueType();
16205 if (VT.isFloatingPoint() && VT.getScalarType() != QueryVT)
16206 return false;
16207
16208 unsigned Opcode = Op.getOpcode();
16209 if (Opcode == ISD::FCANONICALIZE)
16210 return true;
16211
16212 if (auto *CFP = dyn_cast<ConstantFPSDNode>(Op)) {
16213 const auto &F = CFP->getValueAPF();
16214 if (F.isNaN() && F.isSignaling())
16215 return false;
16216 if (!F.isDenormal())
16217 return true;
16218
16219 DenormalMode Mode =
16220 DAG.getMachineFunction().getDenormalMode(F.getSemantics());
16221 return Mode == DenormalMode::getIEEE();
16222 }
16223
16224 // If source is a result of another standard FP operation it is already in
16225 // canonical form.
16226 if (MaxDepth == 0)
16227 return false;
16228
16229 switch (Opcode) {
16230 // These will flush denorms if required.
16231 case ISD::FADD:
16232 case ISD::FSUB:
16233 case ISD::FMUL:
16234 case ISD::FCEIL:
16235 case ISD::FFLOOR:
16236 case ISD::FMA:
16237 case ISD::FMAD:
16238 case ISD::FSQRT:
16239 case ISD::FDIV:
16240 case ISD::FREM:
16241 case ISD::FP_ROUND:
16242 case ISD::FP_EXTEND:
16243 case ISD::FP16_TO_FP:
16244 case ISD::FP_TO_FP16:
16245 case ISD::BF16_TO_FP:
16246 case ISD::FP_TO_BF16:
16247 case ISD::FLDEXP:
16248 case AMDGPUISD::FMUL_LEGACY:
16249 case AMDGPUISD::FMAD_FTZ:
16250 case AMDGPUISD::RCP:
16251 case AMDGPUISD::RSQ:
16252 case AMDGPUISD::RSQ_CLAMP:
16253 case AMDGPUISD::RCP_LEGACY:
16254 case AMDGPUISD::RCP_IFLAG:
16255 case AMDGPUISD::LOG:
16256 case AMDGPUISD::EXP:
16257 case AMDGPUISD::DIV_SCALE:
16258 case AMDGPUISD::DIV_FMAS:
16259 case AMDGPUISD::DIV_FIXUP:
16260 case AMDGPUISD::FRACT:
16261 case AMDGPUISD::CVT_PKRTZ_F16_F32:
16262 case AMDGPUISD::CVT_F32_UBYTE0:
16263 case AMDGPUISD::CVT_F32_UBYTE1:
16264 case AMDGPUISD::CVT_F32_UBYTE2:
16265 case AMDGPUISD::CVT_F32_UBYTE3:
16266 case AMDGPUISD::FP_TO_FP16:
16267 case AMDGPUISD::SIN_HW:
16268 case AMDGPUISD::COS_HW:
16269 return true;
16270
16271 // It can/will be lowered or combined as a bit operation.
16272 // Need to check their input recursively to handle.
16273 case ISD::FNEG:
16274 case ISD::FABS:
16275 case ISD::FCOPYSIGN:
16276 return isCanonicalized(DAG, Op.getOperand(0), QueryVT, UserFlags,
16277 MaxDepth - 1);
16278
16279 case ISD::AND:
16280 if (Op.getValueType() == MVT::i32) {
16281 // Be careful as we only know it is a bitcast floating point type. It
16282 // could be f32, v2f16, we have no way of knowing. Luckily the constant
16283 // value that we optimize for, which comes up in fp32 to bf16 conversions,
16284 // is valid to optimize for all types.
16285 if (auto *RHS = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
16286 if (RHS->getZExtValue() == 0xffff0000) {
16287 return isCanonicalized(DAG, Op.getOperand(0), QueryVT, UserFlags,
16288 MaxDepth - 1);
16289 }
16290 }
16291 }
16292 break;
16293
16294 case ISD::FSIN:
16295 case ISD::FCOS:
16296 case ISD::FSINCOS:
16297 return Op.getValueType().getScalarType() != MVT::f16;
16298
16299 case ISD::FMINNUM:
16300 case ISD::FMAXNUM:
16301 case ISD::FMINNUM_IEEE:
16302 case ISD::FMAXNUM_IEEE:
16303 case ISD::FMINIMUM:
16304 case ISD::FMAXIMUM:
16305 case ISD::FMINIMUMNUM:
16306 case ISD::FMAXIMUMNUM:
16307 case AMDGPUISD::CLAMP:
16308 case AMDGPUISD::FMED3:
16309 case AMDGPUISD::FMAX3:
16310 case AMDGPUISD::FMIN3:
16311 case AMDGPUISD::FMAXIMUM3:
16312 case AMDGPUISD::FMINIMUM3: {
16313 // FIXME: Shouldn't treat the generic operations different based these.
16314 // However, we aren't really required to flush the result from
16315 // minnum/maxnum..
16316
16317 // snans will be quieted, so we only need to worry about denormals.
16318 if (Subtarget->supportsMinMaxDenormModes() ||
16319 // FIXME: denormalsEnabledForType is broken for dynamic
16320 denormalsEnabledForType(DAG, Op.getValueType()))
16321 return true;
16322
16323 // Flushing may be required.
16324 // In pre-GFX9 targets V_MIN_F32 and others do not flush denorms. For such
16325 // targets need to check their input recursively.
16326
16327 // FIXME: Does this apply with clamp? It's implemented with max.
16328 for (unsigned I = 0, E = Op.getNumOperands(); I != E; ++I) {
16329 if (!isCanonicalized(DAG, Op.getOperand(I), QueryVT, UserFlags,
16330 MaxDepth - 1))
16331 return false;
16332 }
16333
16334 return true;
16335 }
16336 case ISD::SELECT: {
16337 return isCanonicalized(DAG, Op.getOperand(1), QueryVT, UserFlags,
16338 MaxDepth - 1) &&
16339 isCanonicalized(DAG, Op.getOperand(2), QueryVT, UserFlags,
16340 MaxDepth - 1);
16341 }
16342 case ISD::BUILD_VECTOR: {
16343 for (unsigned i = 0, e = Op.getNumOperands(); i != e; ++i) {
16344 SDValue SrcOp = Op.getOperand(i);
16345 if (!isCanonicalized(DAG, SrcOp, QueryVT, UserFlags, MaxDepth - 1))
16346 return false;
16347 }
16348
16349 return true;
16350 }
16353 return isCanonicalized(DAG, Op.getOperand(0), QueryVT, UserFlags,
16354 MaxDepth - 1);
16355 }
16357 return isCanonicalized(DAG, Op.getOperand(0), QueryVT, UserFlags,
16358 MaxDepth - 1) &&
16359 isCanonicalized(DAG, Op.getOperand(1), QueryVT, UserFlags,
16360 MaxDepth - 1);
16361 }
16362 case ISD::POISON:
16363 return true;
16364 case ISD::UNDEF:
16365 // Could be anything.
16366 return false;
16367
16368 case ISD::BITCAST: {
16369 // Carry QueryVT through the bitcast unchanged. The top-of-function guard
16370 // rejects a source whose FP format differs from the consumed type, so a
16371 // value canonical in one FP format is not assumed canonical in another.
16372 SDValue Src = peekThroughBitcasts(Op.getOperand(0));
16373 return isCanonicalized(DAG, Src, QueryVT, UserFlags, MaxDepth - 1);
16374 }
16375 case ISD::TRUNCATE: {
16376 // Hack round the mess we make when legalizing extract_vector_elt
16377 if (Op.getValueType() == MVT::i16) {
16378 SDValue TruncSrc = Op.getOperand(0);
16379 if (TruncSrc.getValueType() == MVT::i32 &&
16380 TruncSrc.getOpcode() == ISD::BITCAST &&
16381 TruncSrc.getOperand(0).getValueType() == MVT::v2f16) {
16382 return isCanonicalized(DAG, TruncSrc.getOperand(0), QueryVT, UserFlags,
16383 MaxDepth - 1);
16384 }
16385 }
16386 return false;
16387 }
16389 unsigned IntrinsicID = Op.getConstantOperandVal(0);
16390 // TODO: Handle more intrinsics
16391 switch (IntrinsicID) {
16392 case Intrinsic::amdgcn_cvt_pkrtz:
16393 case Intrinsic::amdgcn_cubeid:
16394 case Intrinsic::amdgcn_frexp_mant:
16395 case Intrinsic::amdgcn_fdot2:
16396 case Intrinsic::amdgcn_rcp:
16397 case Intrinsic::amdgcn_rsq:
16398 case Intrinsic::amdgcn_rsq_clamp:
16399 case Intrinsic::amdgcn_rcp_legacy:
16400 case Intrinsic::amdgcn_rsq_legacy:
16401 case Intrinsic::amdgcn_trig_preop:
16402 case Intrinsic::amdgcn_tanh:
16403 case Intrinsic::amdgcn_log:
16404 case Intrinsic::amdgcn_exp2:
16405 case Intrinsic::amdgcn_sqrt:
16406 return true;
16407 default:
16408 break;
16409 }
16410
16411 break;
16412 }
16413 default:
16414 break;
16415 }
16416
16417 // FIXME: denormalsEnabledForType is broken for dynamic
16418 return denormalsEnabledForType(DAG, Op.getValueType()) &&
16419 (UserFlags.hasNoNaNs() || DAG.isKnownNeverSNaN(Op));
16420}
16421
16423 unsigned MaxDepth) const {
16424 const MachineRegisterInfo &MRI = MF.getRegInfo();
16425 MachineInstr *MI = MRI.getVRegDef(Reg);
16426 unsigned Opcode = MI->getOpcode();
16427
16428 if (Opcode == AMDGPU::G_FCANONICALIZE)
16429 return true;
16430
16431 std::optional<FPValueAndVReg> FCR;
16432 // Constant splat (can be padded with undef) or scalar constant.
16433 if (mi_match(Reg, MRI, MIPatternMatch::m_GFCstOrSplat(FCR))) {
16434 if (FCR->Value.isSignaling())
16435 return false;
16436 if (!FCR->Value.isDenormal())
16437 return true;
16438
16439 DenormalMode Mode = MF.getDenormalMode(FCR->Value.getSemantics());
16440 return Mode == DenormalMode::getIEEE();
16441 }
16442
16443 if (MaxDepth == 0)
16444 return false;
16445
16446 switch (Opcode) {
16447 case AMDGPU::G_FADD:
16448 case AMDGPU::G_FSUB:
16449 case AMDGPU::G_FMUL:
16450 case AMDGPU::G_FCEIL:
16451 case AMDGPU::G_FFLOOR:
16452 case AMDGPU::G_FRINT:
16453 case AMDGPU::G_FNEARBYINT:
16454 case AMDGPU::G_INTRINSIC_FPTRUNC_ROUND:
16455 case AMDGPU::G_INTRINSIC_TRUNC:
16456 case AMDGPU::G_INTRINSIC_ROUNDEVEN:
16457 case AMDGPU::G_FMA:
16458 case AMDGPU::G_FMAD:
16459 case AMDGPU::G_FSQRT:
16460 case AMDGPU::G_FDIV:
16461 case AMDGPU::G_FREM:
16462 case AMDGPU::G_FPOW:
16463 case AMDGPU::G_FPEXT:
16464 case AMDGPU::G_FLOG:
16465 case AMDGPU::G_FLOG2:
16466 case AMDGPU::G_FLOG10:
16467 case AMDGPU::G_FPTRUNC:
16468 case AMDGPU::G_AMDGPU_RCP_IFLAG:
16469 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE0:
16470 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE1:
16471 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE2:
16472 case AMDGPU::G_AMDGPU_CVT_F32_UBYTE3:
16473 return true;
16474 case AMDGPU::G_FNEG:
16475 case AMDGPU::G_FABS:
16476 case AMDGPU::G_FCOPYSIGN:
16477 return isCanonicalized(MI->getOperand(1).getReg(), MF, MaxDepth - 1);
16478 case AMDGPU::G_FMINNUM:
16479 case AMDGPU::G_FMAXNUM:
16480 case AMDGPU::G_FMINNUM_IEEE:
16481 case AMDGPU::G_FMAXNUM_IEEE:
16482 case AMDGPU::G_FMINIMUM:
16483 case AMDGPU::G_FMAXIMUM:
16484 case AMDGPU::G_FMINIMUMNUM:
16485 case AMDGPU::G_FMAXIMUMNUM: {
16486 if (Subtarget->supportsMinMaxDenormModes() ||
16487 // FIXME: denormalsEnabledForType is broken for dynamic
16488 denormalsEnabledForType(MRI.getType(Reg), MF))
16489 return true;
16490
16491 [[fallthrough]];
16492 }
16493 case AMDGPU::G_BUILD_VECTOR:
16494 for (const MachineOperand &MO : llvm::drop_begin(MI->operands()))
16495 if (!isCanonicalized(MO.getReg(), MF, MaxDepth - 1))
16496 return false;
16497 return true;
16498 case AMDGPU::G_INTRINSIC:
16499 case AMDGPU::G_INTRINSIC_CONVERGENT:
16500 switch (cast<GIntrinsic>(MI)->getIntrinsicID()) {
16501 case Intrinsic::amdgcn_fmul_legacy:
16502 case Intrinsic::amdgcn_fmad_ftz:
16503 case Intrinsic::amdgcn_sqrt:
16504 case Intrinsic::amdgcn_fmed3:
16505 case Intrinsic::amdgcn_sin:
16506 case Intrinsic::amdgcn_cos:
16507 case Intrinsic::amdgcn_log:
16508 case Intrinsic::amdgcn_exp2:
16509 case Intrinsic::amdgcn_log_clamp:
16510 case Intrinsic::amdgcn_rcp:
16511 case Intrinsic::amdgcn_rcp_legacy:
16512 case Intrinsic::amdgcn_rsq:
16513 case Intrinsic::amdgcn_rsq_clamp:
16514 case Intrinsic::amdgcn_rsq_legacy:
16515 case Intrinsic::amdgcn_div_scale:
16516 case Intrinsic::amdgcn_div_fmas:
16517 case Intrinsic::amdgcn_div_fixup:
16518 case Intrinsic::amdgcn_fract:
16519 case Intrinsic::amdgcn_cvt_pkrtz:
16520 case Intrinsic::amdgcn_cubeid:
16521 case Intrinsic::amdgcn_cubema:
16522 case Intrinsic::amdgcn_cubesc:
16523 case Intrinsic::amdgcn_cubetc:
16524 case Intrinsic::amdgcn_frexp_mant:
16525 case Intrinsic::amdgcn_fdot2:
16526 case Intrinsic::amdgcn_trig_preop:
16527 case Intrinsic::amdgcn_tanh:
16528 return true;
16529 default:
16530 break;
16531 }
16532
16533 [[fallthrough]];
16534 default:
16535 return false;
16536 }
16537
16538 llvm_unreachable("invalid operation");
16539}
16540
16541// Constant fold canonicalize.
16542SDValue SITargetLowering::getCanonicalConstantFP(SelectionDAG &DAG,
16543 const SDLoc &SL, EVT VT,
16544 const APFloat &C) const {
16545 // Flush denormals to 0 if not enabled.
16546 if (C.isDenormal()) {
16547 DenormalMode Mode =
16548 DAG.getMachineFunction().getDenormalMode(C.getSemantics());
16549 if (Mode == DenormalMode::getPreserveSign()) {
16550 return DAG.getConstantFP(
16551 APFloat::getZero(C.getSemantics(), C.isNegative()), SL, VT);
16552 }
16553
16554 if (Mode != DenormalMode::getIEEE())
16555 return SDValue();
16556 }
16557
16558 if (C.isNaN()) {
16559 if (C.isSignaling()) {
16560 // Quiet a signaling NaN.
16561 return DAG.getConstantFP(C.makeQuiet(), SL, VT);
16562 }
16563 }
16564
16565 // Already canonical.
16566 return DAG.getConstantFP(C, SL, VT);
16567}
16568
16570 return Op.isUndef() || isa<ConstantFPSDNode>(Op);
16571}
16572
16573SDValue
16574SITargetLowering::performFCanonicalizeCombine(SDNode *N,
16575 DAGCombinerInfo &DCI) const {
16576 SelectionDAG &DAG = DCI.DAG;
16577 SDValue N0 = N->getOperand(0);
16578 EVT VT = N->getValueType(0);
16579
16580 // fcanonicalize undef -> qnan
16581 if (N0.isUndef()) {
16583 return DAG.getConstantFP(QNaN, SDLoc(N), VT);
16584 }
16585
16586 if (ConstantFPSDNode *CFP = isConstOrConstSplatFP(N0))
16587 return getCanonicalConstantFP(DAG, SDLoc(N), VT, CFP->getValueAPF());
16588
16589 // fcanonicalize (build_vector x, k) -> build_vector (fcanonicalize x),
16590 // (fcanonicalize k)
16591 //
16592 // fcanonicalize (build_vector x, undef) -> build_vector (fcanonicalize x), 0
16593
16594 // TODO: This could be better with wider vectors that will be split to v2f16,
16595 // and to consider uses since there aren't that many packed operations.
16596 if (N0.getOpcode() == ISD::BUILD_VECTOR && N0.getNumOperands() == 2 &&
16597 isTypeLegal(VT)) {
16598 SDLoc SL(N);
16599 SDValue NewElts[2];
16600 SDValue Lo = N0.getOperand(0);
16601 SDValue Hi = N0.getOperand(1);
16602 EVT EltVT = Lo.getValueType();
16603
16604 // Only apply this optimization if scalar canonicalize is legal for the
16605 // element type. Otherwise, scalarizing may require widening the scalar back
16606 // to a vector, adding overhead (e.g., bf16 has no scalar instructions).
16608 return SDValue();
16609
16611 for (unsigned I = 0; I != 2; ++I) {
16612 SDValue Op = N0.getOperand(I);
16613 if (ConstantFPSDNode *CFP = dyn_cast<ConstantFPSDNode>(Op)) {
16614 NewElts[I] =
16615 getCanonicalConstantFP(DAG, SL, EltVT, CFP->getValueAPF());
16616 } else if (Op.isUndef()) {
16617 // Handled below based on what the other operand is.
16618 NewElts[I] = Op;
16619 } else {
16620 NewElts[I] = DAG.getNode(ISD::FCANONICALIZE, SL, EltVT, Op);
16621 }
16622 }
16623
16624 // If one half is undef, and one is constant, prefer a splat vector.
16625 // Otherwise, convert the undef to 0.0 since that's cheaper to use and may
16626 // be free with a packed operation.
16627 if (NewElts[0].isUndef()) {
16628 NewElts[0] = isa<ConstantFPSDNode>(NewElts[1])
16629 ? NewElts[1]
16630 : DAG.getConstantFP(0.0f, SL, EltVT);
16631 }
16632
16633 if (NewElts[1].isUndef()) {
16634 NewElts[1] = isa<ConstantFPSDNode>(NewElts[0])
16635 ? NewElts[0]
16636 : DAG.getConstantFP(0.0f, SL, EltVT);
16637 }
16638
16639 return DAG.getBuildVector(VT, SL, NewElts);
16640 }
16641 }
16642
16643 return SDValue();
16644}
16645
16646static unsigned minMaxOpcToMin3Max3Opc(unsigned Opc) {
16647 switch (Opc) {
16648 case ISD::FMAXNUM:
16649 case ISD::FMAXNUM_IEEE:
16650 case ISD::FMAXIMUMNUM:
16651 return AMDGPUISD::FMAX3;
16652 case ISD::FMAXIMUM:
16653 return AMDGPUISD::FMAXIMUM3;
16654 case ISD::SMAX:
16655 return AMDGPUISD::SMAX3;
16656 case ISD::UMAX:
16657 return AMDGPUISD::UMAX3;
16658 case ISD::FMINNUM:
16659 case ISD::FMINNUM_IEEE:
16660 case ISD::FMINIMUMNUM:
16661 return AMDGPUISD::FMIN3;
16662 case ISD::FMINIMUM:
16663 return AMDGPUISD::FMINIMUM3;
16664 case ISD::SMIN:
16665 return AMDGPUISD::SMIN3;
16666 case ISD::UMIN:
16667 return AMDGPUISD::UMIN3;
16668 default:
16669 llvm_unreachable("Not a min/max opcode");
16670 }
16671}
16672
16673SDValue SITargetLowering::performIntMed3ImmCombine(SelectionDAG &DAG,
16674 const SDLoc &SL, SDValue Src,
16675 SDValue MinVal,
16676 SDValue MaxVal,
16677 bool Signed) const {
16678
16679 // med3 comes from
16680 // min(max(x, K0), K1), K0 < K1
16681 // max(min(x, K0), K1), K1 < K0
16682 //
16683 // "MinVal" and "MaxVal" respectively refer to the rhs of the
16684 // min/max op.
16685 ConstantSDNode *MinK = dyn_cast<ConstantSDNode>(MinVal);
16686 ConstantSDNode *MaxK = dyn_cast<ConstantSDNode>(MaxVal);
16687
16688 if (!MinK || !MaxK)
16689 return SDValue();
16690
16691 if (Signed) {
16692 if (MaxK->getAPIntValue().sge(MinK->getAPIntValue()))
16693 return SDValue();
16694 } else {
16695 if (MaxK->getAPIntValue().uge(MinK->getAPIntValue()))
16696 return SDValue();
16697 }
16698
16699 EVT VT = MinK->getValueType(0);
16700 unsigned Med3Opc = Signed ? AMDGPUISD::SMED3 : AMDGPUISD::UMED3;
16701 if (VT == MVT::i32 || (VT == MVT::i16 && Subtarget->hasMed3_16()))
16702 return DAG.getNode(Med3Opc, SL, VT, Src, MaxVal, MinVal);
16703
16704 // Note: we could also extend to i32 and use i32 med3 if i16 med3 is
16705 // not available, but this is unlikely to be profitable as constants
16706 // will often need to be materialized & extended, especially on
16707 // pre-GFX10 where VOP3 instructions couldn't take literal operands.
16708 return SDValue();
16709}
16710
16713 return C;
16714
16716 if (ConstantFPSDNode *C = BV->getConstantFPSplatNode())
16717 return C;
16718 }
16719
16720 return nullptr;
16721}
16722
16723SDValue SITargetLowering::performFPMed3ImmCombine(SelectionDAG &DAG,
16724 const SDLoc &SL, SDValue Op0,
16725 SDValue Op1,
16726 bool IsKnownNoNaNs) const {
16727 ConstantFPSDNode *K1 = getSplatConstantFP(Op1);
16728 if (!K1)
16729 return SDValue();
16730
16731 ConstantFPSDNode *K0 = getSplatConstantFP(Op0.getOperand(1));
16732 if (!K0)
16733 return SDValue();
16734
16735 // Ordered >= (although NaN inputs should have folded away by now).
16736 if (K0->getValueAPF() > K1->getValueAPF())
16737 return SDValue();
16738
16739 // med3 with a nan input acts like
16740 // v_min_f32(v_min_f32(S0.f32, S1.f32), S2.f32)
16741 //
16742 // So the result depends on whether the IEEE mode bit is enabled or not with a
16743 // signaling nan input.
16744 // ieee=1
16745 // s0 snan: yields s2
16746 // s1 snan: yields s2
16747 // s2 snan: qnan
16748
16749 // s0 qnan: min(s1, s2)
16750 // s1 qnan: min(s0, s2)
16751 // s2 qnan: min(s0, s1)
16752
16753 // ieee=0
16754 // s0 snan: min(s1, s2)
16755 // s1 snan: min(s0, s2)
16756 // s2 snan: qnan
16757
16758 // s0 qnan: min(s1, s2)
16759 // s1 qnan: min(s0, s2)
16760 // s2 qnan: min(s0, s1)
16761 const MachineFunction &MF = DAG.getMachineFunction();
16762 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
16763
16764 // TODO: Check IEEE bit enabled. We can form fmed3 with IEEE=0 regardless of
16765 // whether the input is a signaling nan if op0 is fmaximum or fmaximumnum. We
16766 // can only form if op0 is fmaxnum_ieee if IEEE=1.
16767 EVT VT = Op0.getValueType();
16768 if (Info->getMode().DX10Clamp) {
16769 // If dx10_clamp is enabled, NaNs clamp to 0.0. This is the same as the
16770 // hardware fmed3 behavior converting to a min.
16771 // FIXME: Should this be allowing -0.0?
16772 if (K1->isOne() && K0->isPosZero())
16773 return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Op0.getOperand(0));
16774 }
16775
16776 // med3 for f16 is only available on gfx9+, and not available for v2f16.
16777 if (VT == MVT::f32 || (VT == MVT::f16 && Subtarget->hasMed3_16())) {
16778 // This isn't safe with signaling NaNs because in IEEE mode, min/max on a
16779 // signaling NaN gives a quiet NaN. The quiet NaN input to the min would
16780 // then give the other result, which is different from med3 with a NaN
16781 // input.
16782 SDValue Var = Op0.getOperand(0);
16783 if (!IsKnownNoNaNs && !DAG.isKnownNeverSNaN(Var))
16784 return SDValue();
16785
16786 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
16787
16788 if ((!K0->hasOneUse() || TII->isInlineConstant(K0->getValueAPF())) &&
16789 (!K1->hasOneUse() || TII->isInlineConstant(K1->getValueAPF()))) {
16790 return DAG.getNode(AMDGPUISD::FMED3, SL, K0->getValueType(0), Var,
16791 SDValue(K0, 0), SDValue(K1, 0));
16792 }
16793 }
16794
16795 return SDValue();
16796}
16797
16798/// \return true if the subtarget supports minimum3 and maximum3 with the given
16799/// base min/max opcode \p Opc for type \p VT.
16800static bool supportsMin3Max3(const GCNSubtarget &Subtarget, unsigned Opc,
16801 EVT VT) {
16802 switch (Opc) {
16803 case ISD::FMINNUM:
16804 case ISD::FMAXNUM:
16805 case ISD::FMINNUM_IEEE:
16806 case ISD::FMAXNUM_IEEE:
16807 case ISD::FMINIMUMNUM:
16808 case ISD::FMAXIMUMNUM:
16809 case AMDGPUISD::FMIN_LEGACY:
16810 case AMDGPUISD::FMAX_LEGACY:
16811 return (VT == MVT::f32) || (VT == MVT::f16 && Subtarget.hasMin3Max3_16()) ||
16812 (VT == MVT::v2f16 && Subtarget.hasMin3Max3PKF16());
16813 case ISD::FMINIMUM:
16814 case ISD::FMAXIMUM:
16815 return (VT == MVT::f32 && Subtarget.hasMinimum3Maximum3F32()) ||
16816 (VT == MVT::f16 && Subtarget.hasMinimum3Maximum3F16()) ||
16817 (VT == MVT::v2f16 && Subtarget.hasMinimum3Maximum3PKF16());
16818 case ISD::SMAX:
16819 case ISD::SMIN:
16820 case ISD::UMAX:
16821 case ISD::UMIN:
16822 return (VT == MVT::i32) || (VT == MVT::i16 && Subtarget.hasMin3Max3_16());
16823 default:
16824 return false;
16825 }
16826
16827 llvm_unreachable("not a min/max opcode");
16828}
16829
16830SDValue SITargetLowering::performMinMaxCombine(SDNode *N,
16831 DAGCombinerInfo &DCI) const {
16832 SelectionDAG &DAG = DCI.DAG;
16833
16834 EVT VT = N->getValueType(0);
16835 unsigned Opc = N->getOpcode();
16836 SDValue Op0 = N->getOperand(0);
16837 SDValue Op1 = N->getOperand(1);
16838
16839 // Only do this if the inner op has one use since this will just increases
16840 // register pressure for no benefit.
16841
16842 if (supportsMin3Max3(*Subtarget, Opc, VT)) {
16843 auto IsTreeWithCombinableChildren = [Opc](SDValue Op) {
16844 return (Op.getOperand(0).getOpcode() == Opc &&
16845 Op.getOperand(0).hasOneUse()) ||
16846 (Op.getOperand(1).getOpcode() == Opc &&
16847 Op.getOperand(1).hasOneUse());
16848 };
16849
16850 bool CanTreeCombineApply = Op0.getOpcode() == Opc && Op0.hasOneUse() &&
16851 Op1.getOpcode() == Opc && Op1.hasOneUse();
16852 bool HasCombinableTreeChild =
16853 CanTreeCombineApply && (IsTreeWithCombinableChildren(Op0) ||
16854 IsTreeWithCombinableChildren(Op1));
16855
16856 // Tree reduction: when both operands are the same min/max op, restructure
16857 // to keep a 2-op node on top so higher tree levels can still combine.
16858 //
16859 // max(max(a, b), max(c, d)) -> max(max3(a, b, c), d)
16860 // min(min(a, b), min(c, d)) -> min(min3(a, b, c), d)
16861 //
16862 // Defer when either inner op is a tree node with combinable children.
16863 if (CanTreeCombineApply && !HasCombinableTreeChild) {
16864 SDLoc DL(N);
16865 SDValue Inner =
16867 Op0.getOperand(1), Op1.getOperand(0));
16868 return DAG.getNode(Opc, DL, VT, Inner, Op1.getOperand(1));
16869 }
16870
16871 // max(max(a, b), c) -> max3(a, b, c)
16872 // min(min(a, b), c) -> min3(a, b, c)
16873 // Deferred when Op0 is a tree node with combinable children.
16874 if (Op0.getOpcode() == Opc && Op0.hasOneUse() && !HasCombinableTreeChild) {
16875 SDLoc DL(N);
16876 return DAG.getNode(minMaxOpcToMin3Max3Opc(Opc), DL, N->getValueType(0),
16877 Op0.getOperand(0), Op0.getOperand(1), Op1);
16878 }
16879
16880 // Try commuted.
16881 // max(a, max(b, c)) -> max3(a, b, c)
16882 // min(a, min(b, c)) -> min3(a, b, c)
16883 // Deferred when Op1 is a tree node with combinable children.
16884 if (Op1.getOpcode() == Opc && Op1.hasOneUse() && !HasCombinableTreeChild) {
16885 SDLoc DL(N);
16886 return DAG.getNode(minMaxOpcToMin3Max3Opc(Opc), DL, N->getValueType(0),
16887 Op0, Op1.getOperand(0), Op1.getOperand(1));
16888 }
16889 }
16890
16891 // umin(sffbh(x), bitwidth) -> sffbh(x) if x is known to be not 0 or -1.
16892 SDValue FfbhSrc;
16893 uint64_t Clamp = 0;
16894 if (Opc == ISD::UMIN &&
16895 sd_match(Op0,
16897 sd_match(Op1, m_ConstInt(Clamp))) {
16898 unsigned BitWidth = FfbhSrc.getValueType().getScalarSizeInBits();
16899 if (Clamp >= BitWidth) {
16900 KnownBits Known = DAG.computeKnownBits(FfbhSrc);
16901 if (Known.isNonZero() && Known.Zero.getBoolValue())
16902 return Op0;
16903 }
16904 }
16905
16906 // min(max(x, K0), K1), K0 < K1 -> med3(x, K0, K1)
16907 // max(min(x, K0), K1), K1 < K0 -> med3(x, K1, K0)
16908 if (Opc == ISD::SMIN && Op0.getOpcode() == ISD::SMAX && Op0.hasOneUse()) {
16909 if (SDValue Med3 = performIntMed3ImmCombine(
16910 DAG, SDLoc(N), Op0->getOperand(0), Op1, Op0->getOperand(1), true))
16911 return Med3;
16912 }
16913 if (Opc == ISD::SMAX && Op0.getOpcode() == ISD::SMIN && Op0.hasOneUse()) {
16914 if (SDValue Med3 = performIntMed3ImmCombine(
16915 DAG, SDLoc(N), Op0->getOperand(0), Op0->getOperand(1), Op1, true))
16916 return Med3;
16917 }
16918
16919 if (Opc == ISD::UMIN && Op0.getOpcode() == ISD::UMAX && Op0.hasOneUse()) {
16920 if (SDValue Med3 = performIntMed3ImmCombine(
16921 DAG, SDLoc(N), Op0->getOperand(0), Op1, Op0->getOperand(1), false))
16922 return Med3;
16923 }
16924 if (Opc == ISD::UMAX && Op0.getOpcode() == ISD::UMIN && Op0.hasOneUse()) {
16925 if (SDValue Med3 = performIntMed3ImmCombine(
16926 DAG, SDLoc(N), Op0->getOperand(0), Op0->getOperand(1), Op1, false))
16927 return Med3;
16928 }
16929
16930 // if !is_snan(x):
16931 // fminnum(fmaxnum(x, K0), K1), K0 < K1 -> fmed3(x, K0, K1)
16932 // fminnum_ieee(fmaxnum_ieee(x, K0), K1), K0 < K1 -> fmed3(x, K0, K1)
16933 // fminnumnum(fmaxnumnum(x, K0), K1), K0 < K1 -> fmed3(x, K0, K1)
16934 // fmin_legacy(fmax_legacy(x, K0), K1), K0 < K1 -> fmed3(x, K0, K1)
16935 if (((Opc == ISD::FMINNUM && Op0.getOpcode() == ISD::FMAXNUM) ||
16938 (Opc == AMDGPUISD::FMIN_LEGACY &&
16939 Op0.getOpcode() == AMDGPUISD::FMAX_LEGACY)) &&
16940 (VT == MVT::f32 || VT == MVT::f64 ||
16941 (VT == MVT::f16 && Subtarget->has16BitInsts()) ||
16942 (VT == MVT::bf16 && Subtarget->hasBF16PackedInsts()) ||
16943 (VT == MVT::v2bf16 && Subtarget->hasBF16PackedInsts()) ||
16944 (VT == MVT::v2f16 && Subtarget->hasVOP3PInsts())) &&
16945 Op0.hasOneUse()) {
16946 if (SDValue Res = performFPMed3ImmCombine(DAG, SDLoc(N), Op0, Op1,
16947 N->getFlags().hasNoNaNs()))
16948 return Res;
16949 }
16950
16951 // Prefer fminnum_ieee over fminimum. For gfx950, minimum/maximum are legal
16952 // for some types, but at a higher cost since it's implemented with a 3
16953 // operand form.
16954 const SDNodeFlags Flags = N->getFlags();
16955 if ((Opc == ISD::FMINIMUM || Opc == ISD::FMAXIMUM) && Flags.hasNoNaNs() &&
16956 !Subtarget->hasIEEEMinimumMaximumInsts() &&
16958 unsigned NewOpc =
16960 return DAG.getNode(NewOpc, SDLoc(N), VT, Op0, Op1, Flags);
16961 }
16962
16963 return SDValue();
16964}
16965
16969 // FIXME: Should this be allowing -0.0?
16970 return (CA->isPosZero() && CB->isOne()) ||
16971 (CA->isOne() && CB->isPosZero());
16972 }
16973 }
16974
16975 return false;
16976}
16977
16978// FIXME: Should only worry about snans for version with chain.
16979SDValue SITargetLowering::performFMed3Combine(SDNode *N,
16980 DAGCombinerInfo &DCI) const {
16981 EVT VT = N->getValueType(0);
16982 // v_med3_f32 and v_max_f32 behave identically wrt denorms, exceptions and
16983 // NaNs. With a NaN input, the order of the operands may change the result.
16984
16985 SelectionDAG &DAG = DCI.DAG;
16986 SDLoc SL(N);
16987
16988 SDValue Src0 = N->getOperand(0);
16989 SDValue Src1 = N->getOperand(1);
16990 SDValue Src2 = N->getOperand(2);
16991
16992 if (isClampZeroToOne(Src0, Src1)) {
16993 // const_a, const_b, x -> clamp is safe in all cases including signaling
16994 // nans.
16995 // FIXME: Should this be allowing -0.0?
16996 return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Src2);
16997 }
16998
16999 const MachineFunction &MF = DAG.getMachineFunction();
17000 const SIMachineFunctionInfo *Info = MF.getInfo<SIMachineFunctionInfo>();
17001
17002 // FIXME: dx10_clamp behavior assumed in instcombine. Should we really bother
17003 // handling no dx10-clamp?
17004 if (Info->getMode().DX10Clamp) {
17005 // If NaNs is clamped to 0, we are free to reorder the inputs.
17006
17007 if (isa<ConstantFPSDNode>(Src0) && !isa<ConstantFPSDNode>(Src1))
17008 std::swap(Src0, Src1);
17009
17010 if (isa<ConstantFPSDNode>(Src1) && !isa<ConstantFPSDNode>(Src2))
17011 std::swap(Src1, Src2);
17012
17013 if (isa<ConstantFPSDNode>(Src0) && !isa<ConstantFPSDNode>(Src1))
17014 std::swap(Src0, Src1);
17015
17016 if (isClampZeroToOne(Src1, Src2))
17017 return DAG.getNode(AMDGPUISD::CLAMP, SL, VT, Src0);
17018 }
17019
17020 return SDValue();
17021}
17022
17023SDValue SITargetLowering::performCvtPkRTZCombine(SDNode *N,
17024 DAGCombinerInfo &DCI) const {
17025 SDValue Src0 = N->getOperand(0);
17026 SDValue Src1 = N->getOperand(1);
17027 if (Src0.isUndef() && Src1.isUndef())
17028 return DCI.DAG.getUNDEF(N->getValueType(0));
17029 return SDValue();
17030}
17031
17032// Check if EXTRACT_VECTOR_ELT/INSERT_VECTOR_ELT (<n x e>, var-idx) should be
17033// expanded into a set of cmp/select instructions.
17035 unsigned NumElem,
17036 bool IsDivergentIdx,
17037 const GCNSubtarget *Subtarget) {
17039 return false;
17040
17041 unsigned VecSize = EltSize * NumElem;
17042
17043 // Sub-dword vectors of size 2 dword or less have better implementation.
17044 if (VecSize <= 64 && EltSize < 32)
17045 return false;
17046
17047 // Always expand the rest of sub-dword instructions, otherwise it will be
17048 // lowered via memory.
17049 if (EltSize < 32)
17050 return true;
17051
17052 // Always do this if var-idx is divergent, otherwise it will become a loop.
17053 if (IsDivergentIdx)
17054 return true;
17055
17056 // Large vectors would yield too many compares and v_cndmask_b32 instructions.
17057 unsigned NumInsts = NumElem /* Number of compares */ +
17058 ((EltSize + 31) / 32) * NumElem /* Number of cndmasks */;
17059
17060 // On some architectures (GFX9) movrel is not available and it's better
17061 // to expand.
17062 if (Subtarget->useVGPRIndexMode())
17063 return NumInsts <= 16;
17064
17065 // If movrel is available, use it instead of expanding for vector of 8
17066 // elements.
17067 if (Subtarget->hasMovrel())
17068 return NumInsts <= 15;
17069
17070 return true;
17071}
17072
17074 SDValue Idx = N->getOperand(N->getNumOperands() - 1);
17075 if (isa<ConstantSDNode>(Idx))
17076 return false;
17077
17078 SDValue Vec = N->getOperand(0);
17079 EVT VecVT = Vec.getValueType();
17080 EVT EltVT = VecVT.getVectorElementType();
17081 unsigned EltSize = EltVT.getSizeInBits();
17082 unsigned NumElem = VecVT.getVectorNumElements();
17083
17085 EltSize, NumElem, Idx->isDivergent(), getSubtarget());
17086}
17087
17088SDValue
17089SITargetLowering::performExtractVectorEltCombine(SDNode *N,
17090 DAGCombinerInfo &DCI) const {
17091 SDValue Vec = N->getOperand(0);
17092 SelectionDAG &DAG = DCI.DAG;
17093
17094 EVT VecVT = Vec.getValueType();
17095 EVT VecEltVT = VecVT.getVectorElementType();
17096 EVT ResVT = N->getValueType(0);
17097
17098 unsigned VecSize = VecVT.getSizeInBits();
17099 unsigned VecEltSize = VecEltVT.getSizeInBits();
17100
17101 if ((Vec.getOpcode() == ISD::FNEG || Vec.getOpcode() == ISD::FABS) &&
17103 SDLoc SL(N);
17104 SDValue Idx = N->getOperand(1);
17105 SDValue Elt =
17106 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, ResVT, Vec.getOperand(0), Idx);
17107 return DAG.getNode(Vec.getOpcode(), SL, ResVT, Elt);
17108 }
17109
17110 // (extract_vector_element (and {y0, y1}, (build_vector 0x1f, 0x1f)), index)
17111 // -> (and (extract_vector_element {y0, y1}, index), 0x1f)
17112 // There are optimisations to transform 64-bit shifts into 32-bit shifts
17113 // depending on the shift operand. See e.g. performSraCombine().
17114 // This combine ensures that the optimisation is compatible with v2i32
17115 // legalised AND.
17116 if (VecVT == MVT::v2i32 && Vec->getOpcode() == ISD::AND &&
17117 Vec->getOperand(1)->getOpcode() == ISD::BUILD_VECTOR) {
17118
17120 if (!C || C->getZExtValue() != 0x1f)
17121 return SDValue();
17122
17123 SDLoc SL(N);
17124 SDValue AndMask = DAG.getConstant(0x1f, SL, MVT::i32);
17125 SDValue EVE = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32,
17126 Vec->getOperand(0), N->getOperand(1));
17127 SDValue A = DAG.getNode(ISD::AND, SL, MVT::i32, EVE, AndMask);
17128 DAG.ReplaceAllUsesWith(N, A.getNode());
17129 }
17130
17131 // ScalarRes = EXTRACT_VECTOR_ELT ((vector-BINOP Vec1, Vec2), Idx)
17132 // =>
17133 // Vec1Elt = EXTRACT_VECTOR_ELT(Vec1, Idx)
17134 // Vec2Elt = EXTRACT_VECTOR_ELT(Vec2, Idx)
17135 // ScalarRes = scalar-BINOP Vec1Elt, Vec2Elt
17136 if (Vec.hasOneUse() && DCI.isBeforeLegalize() && VecEltVT == ResVT) {
17137 SDLoc SL(N);
17138 SDValue Idx = N->getOperand(1);
17139 unsigned Opc = Vec.getOpcode();
17140
17141 switch (Opc) {
17142 default:
17143 break;
17144 // TODO: Support other binary operations.
17145 case ISD::FADD:
17146 case ISD::FSUB:
17147 case ISD::FMUL:
17148 case ISD::ADD:
17149 case ISD::UMIN:
17150 case ISD::UMAX:
17151 case ISD::SMIN:
17152 case ISD::SMAX:
17153 case ISD::FMAXNUM:
17154 case ISD::FMINNUM:
17155 case ISD::FMAXNUM_IEEE:
17156 case ISD::FMINNUM_IEEE:
17157 case ISD::FMAXIMUM:
17158 case ISD::FMINIMUM: {
17159 SDValue Elt0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, ResVT,
17160 Vec.getOperand(0), Idx);
17161 SDValue Elt1 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, ResVT,
17162 Vec.getOperand(1), Idx);
17163
17164 DCI.AddToWorklist(Elt0.getNode());
17165 DCI.AddToWorklist(Elt1.getNode());
17166 return DAG.getNode(Opc, SL, ResVT, Elt0, Elt1, Vec->getFlags());
17167 }
17168 }
17169 }
17170
17171 // EXTRACT_VECTOR_ELT (<n x e>, var-idx) => n x select (e, const-idx)
17173 SDLoc SL(N);
17174 SDValue Idx = N->getOperand(1);
17175 SDValue V;
17176 for (unsigned I = 0, E = VecVT.getVectorNumElements(); I < E; ++I) {
17177 SDValue IC = DAG.getVectorIdxConstant(I, SL);
17178 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, ResVT, Vec, IC);
17179 if (I == 0)
17180 V = Elt;
17181 else
17182 V = DAG.getSelectCC(SL, Idx, IC, Elt, V, ISD::SETEQ);
17183 }
17184 return V;
17185 }
17186
17187 // EXTRACT_VECTOR_ELT (v2i32 bitcast (i64/f64:k), Idx)
17188 // =>
17189 // i32:Lo(k) if Idx == 0, or
17190 // i32:Hi(k) if Idx == 1
17191 auto *Idx = dyn_cast<ConstantSDNode>(N->getOperand(1));
17192 if (Vec.getOpcode() == ISD::BITCAST && VecVT == MVT::v2i32 && Idx) {
17193 SDLoc SL(N);
17194 SDValue PeekThrough = Vec.getOperand(0);
17195 auto *KImm = dyn_cast<ConstantSDNode>(PeekThrough);
17196 if (KImm && KImm->getValueType(0).getSizeInBits() == 64) {
17197 uint64_t KImmValue = KImm->getZExtValue();
17198 return DAG.getConstant(
17199 (KImmValue >> (32 * Idx->getZExtValue())) & 0xffffffff, SL, MVT::i32);
17200 }
17201 auto *KFPImm = dyn_cast<ConstantFPSDNode>(PeekThrough);
17202 if (KFPImm && KFPImm->getValueType(0).getSizeInBits() == 64) {
17203 uint64_t KFPImmValue =
17204 KFPImm->getValueAPF().bitcastToAPInt().getZExtValue();
17205 return DAG.getConstant((KFPImmValue >> (32 * Idx->getZExtValue())) &
17206 0xffffffff,
17207 SL, MVT::i32);
17208 }
17209 }
17210
17211 if (!DCI.isBeforeLegalize())
17212 return SDValue();
17213
17214 // Try to turn sub-dword accesses of vectors into accesses of the same 32-bit
17215 // elements. This exposes more load reduction opportunities by replacing
17216 // multiple small extract_vector_elements with a single 32-bit extract.
17217 if (isa<MemSDNode>(Vec) && VecEltSize <= 16 && VecEltVT.isByteSized() &&
17218 VecSize > 32 && VecSize % 32 == 0 && Idx) {
17219 EVT NewVT = getEquivalentMemType(*DAG.getContext(), VecVT);
17220
17221 unsigned BitIndex = Idx->getZExtValue() * VecEltSize;
17222 unsigned EltIdx = BitIndex / 32;
17223 unsigned LeftoverBitIdx = BitIndex % 32;
17224 SDLoc SL(N);
17225
17226 SDValue Cast = DAG.getNode(ISD::BITCAST, SL, NewVT, Vec);
17227 DCI.AddToWorklist(Cast.getNode());
17228
17229 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, MVT::i32, Cast,
17230 DAG.getConstant(EltIdx, SL, MVT::i32));
17231 DCI.AddToWorklist(Elt.getNode());
17232 SDValue Srl = DAG.getNode(ISD::SRL, SL, MVT::i32, Elt,
17233 DAG.getConstant(LeftoverBitIdx, SL, MVT::i32));
17234 DCI.AddToWorklist(Srl.getNode());
17235
17236 EVT VecEltAsIntVT = VecEltVT.changeTypeToInteger();
17237 SDValue Trunc = DAG.getNode(ISD::TRUNCATE, SL, VecEltAsIntVT, Srl);
17238 DCI.AddToWorklist(Trunc.getNode());
17239
17240 if (VecEltVT == ResVT) {
17241 return DAG.getNode(ISD::BITCAST, SL, VecEltVT, Trunc);
17242 }
17243
17244 assert(ResVT.isScalarInteger());
17245 return DAG.getAnyExtOrTrunc(Trunc, SL, ResVT);
17246 }
17247
17248 return SDValue();
17249}
17250
17251SDValue
17252SITargetLowering::performInsertVectorEltCombine(SDNode *N,
17253 DAGCombinerInfo &DCI) const {
17254 SDValue Vec = N->getOperand(0);
17255 SDValue Idx = N->getOperand(2);
17256 EVT VecVT = Vec.getValueType();
17257 EVT EltVT = VecVT.getVectorElementType();
17258
17259 // INSERT_VECTOR_ELT (<n x e>, var-idx)
17260 // => BUILD_VECTOR n x select (e, const-idx)
17262 return SDValue();
17263
17264 SelectionDAG &DAG = DCI.DAG;
17265 SDLoc SL(N);
17266 SDValue Ins = N->getOperand(1);
17267 EVT IdxVT = Idx.getValueType();
17268
17270 for (unsigned I = 0, E = VecVT.getVectorNumElements(); I < E; ++I) {
17271 SDValue IC = DAG.getConstant(I, SL, IdxVT);
17272 SDValue Elt = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SL, EltVT, Vec, IC);
17273 SDValue V = DAG.getSelectCC(SL, Idx, IC, Ins, Elt, ISD::SETEQ);
17274 Ops.push_back(V);
17275 }
17276
17277 return DAG.getBuildVector(VecVT, SL, Ops);
17278}
17279
17280/// Return the source of an fp_extend from f16 to f32, or a converted FP
17281/// constant.
17283 if (Src.getOpcode() == ISD::FP_EXTEND &&
17284 Src.getOperand(0).getValueType() == MVT::f16) {
17285 return Src.getOperand(0);
17286 }
17287
17288 if (auto *CFP = dyn_cast<ConstantFPSDNode>(Src)) {
17289 APFloat Val = CFP->getValueAPF();
17290 bool LosesInfo = true;
17292 if (!LosesInfo)
17293 return DAG.getConstantFP(Val, SDLoc(Src), MVT::f16);
17294 }
17295
17296 return SDValue();
17297}
17298
17299SDValue SITargetLowering::performFPRoundCombine(SDNode *N,
17300 DAGCombinerInfo &DCI) const {
17301 assert(Subtarget->has16BitInsts() && !Subtarget->hasMed3_16() &&
17302 "combine only useful on gfx8");
17303
17304 SDValue TruncSrc = N->getOperand(0);
17305 EVT VT = N->getValueType(0);
17306 if (VT != MVT::f16)
17307 return SDValue();
17308
17309 if (TruncSrc.getOpcode() != AMDGPUISD::FMED3 ||
17310 TruncSrc.getValueType() != MVT::f32 || !TruncSrc.hasOneUse())
17311 return SDValue();
17312
17313 SelectionDAG &DAG = DCI.DAG;
17314 SDLoc SL(N);
17315
17316 // Optimize f16 fmed3 pattern performed on f32. On gfx8 there is no f16 fmed3,
17317 // and expanding it with min/max saves 1 instruction vs. casting to f32 and
17318 // casting back.
17319
17320 // fptrunc (f32 (fmed3 (fpext f16:a, fpext f16:b, fpext f16:c))) =>
17321 // fmin(fmax(a, b), fmax(fmin(a, b), c))
17322 SDValue A = strictFPExtFromF16(DAG, TruncSrc.getOperand(0));
17323 if (!A)
17324 return SDValue();
17325
17326 SDValue B = strictFPExtFromF16(DAG, TruncSrc.getOperand(1));
17327 if (!B)
17328 return SDValue();
17329
17330 SDValue C = strictFPExtFromF16(DAG, TruncSrc.getOperand(2));
17331 if (!C)
17332 return SDValue();
17333
17334 // This changes signaling nan behavior. If an input is a signaling nan, it
17335 // would have been quieted by the fpext originally. We don't care because
17336 // these are unconstrained ops. If we needed to insert quieting canonicalizes
17337 // we would be worse off than just doing the promotion.
17338 SDValue A1 = DAG.getNode(ISD::FMINNUM_IEEE, SL, VT, A, B);
17339 SDValue B1 = DAG.getNode(ISD::FMAXNUM_IEEE, SL, VT, A, B);
17340 SDValue C1 = DAG.getNode(ISD::FMAXNUM_IEEE, SL, VT, A1, C);
17341 return DAG.getNode(ISD::FMINNUM_IEEE, SL, VT, B1, C1);
17342}
17343
17344unsigned SITargetLowering::getFusedOpcode(const SelectionDAG &DAG,
17345 const SDNode *N0,
17346 const SDNode *N1) const {
17347 EVT VT = N0->getValueType(0);
17348
17349 // Only do this if we are not trying to support denormals. v_mad_f32 does not
17350 // support denormals ever.
17351 if (((VT == MVT::f32 &&
17353 (VT == MVT::f16 && Subtarget->hasMadF16() &&
17356 return ISD::FMAD;
17357
17358 if (N0->getFlags().hasAllowContract() && N1->getFlags().hasAllowContract() &&
17360 return ISD::FMA;
17361 }
17362
17363 return 0;
17364}
17365
17366// For a reassociatable opcode perform:
17367// op x, (op y, z) -> op (op x, z), y, if x and z are uniform
17368SDValue SITargetLowering::reassociateScalarOps(SDNode *N,
17369 SelectionDAG &DAG) const {
17370 EVT VT = N->getValueType(0);
17371 if (VT != MVT::i32 && VT != MVT::i64)
17372 return SDValue();
17373
17374 if (DAG.isBaseWithConstantOffset(SDValue(N, 0)))
17375 return SDValue();
17376
17377 unsigned Opc = N->getOpcode();
17378 SDValue Op0 = N->getOperand(0);
17379 SDValue Op1 = N->getOperand(1);
17380
17381 if (!(Op0->isDivergent() ^ Op1->isDivergent()))
17382 return SDValue();
17383
17384 if (Op0->isDivergent())
17385 std::swap(Op0, Op1);
17386
17387 if (Op1.getOpcode() != Opc || !Op1.hasOneUse())
17388 return SDValue();
17389
17390 SDValue Op2 = Op1.getOperand(1);
17391 Op1 = Op1.getOperand(0);
17392 if (!(Op1->isDivergent() ^ Op2->isDivergent()))
17393 return SDValue();
17394
17395 if (Op1->isDivergent())
17396 std::swap(Op1, Op2);
17397
17398 SDLoc SL(N);
17399 SDValue Add1 = DAG.getNode(Opc, SL, VT, Op0, Op1);
17400 return DAG.getNode(Opc, SL, VT, Add1, Op2);
17401}
17402
17403static SDValue getMad64_32(SelectionDAG &DAG, const SDLoc &SL, EVT VT,
17404 SDValue N0, SDValue N1, SDValue N2, bool Signed) {
17406 SDVTList VTs = DAG.getVTList(MVT::i64, MVT::i1);
17407 SDValue Mad = DAG.getNode(MadOpc, SL, VTs, N0, N1, N2);
17408 return DAG.getNode(ISD::TRUNCATE, SL, VT, Mad);
17409}
17410
17411// Fold
17412// y = lshr i64 x, 32
17413// res = add (mul i64 y, Const), x where "Const" is a 64-bit constant
17414// with Const.hi == -1
17415// To
17416// res = mad_u64_u32 y.lo ,Const.lo, x.lo
17418 SDValue MulLHS, SDValue MulRHS,
17419 SDValue AddRHS) {
17420 if (MulRHS.getOpcode() == ISD::SRL)
17421 std::swap(MulLHS, MulRHS);
17422
17423 if (MulLHS.getValueType() != MVT::i64 || MulLHS.getOpcode() != ISD::SRL)
17424 return SDValue();
17425
17426 ConstantSDNode *ShiftVal = dyn_cast<ConstantSDNode>(MulLHS.getOperand(1));
17427 if (!ShiftVal || ShiftVal->getAsZExtVal() != 32 ||
17428 MulLHS.getOperand(0) != AddRHS)
17429 return SDValue();
17430
17432 if (!Const || Hi_32(Const->getZExtValue()) != uint32_t(-1))
17433 return SDValue();
17434
17435 SDValue ConstMul =
17436 DAG.getConstant(Lo_32(Const->getZExtValue()), SL, MVT::i32);
17437 return getMad64_32(DAG, SL, MVT::i64,
17438 DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, MulLHS), ConstMul,
17439 DAG.getZeroExtendInReg(AddRHS, SL, MVT::i32), false);
17440}
17441
17442// Fold (add (mul x, y), z) --> (mad_[iu]64_[iu]32 x, y, z) plus high
17443// multiplies, if any.
17444//
17445// Full 64-bit multiplies that feed into an addition are lowered here instead
17446// of using the generic expansion. The generic expansion ends up with
17447// a tree of ADD nodes that prevents us from using the "add" part of the
17448// MAD instruction. The expansion produced here results in a chain of ADDs
17449// instead of a tree.
17450SDValue SITargetLowering::tryFoldToMad64_32(SDNode *N,
17451 DAGCombinerInfo &DCI) const {
17452 assert(N->isAnyAdd());
17453
17454 SelectionDAG &DAG = DCI.DAG;
17455 EVT VT = N->getValueType(0);
17456 SDLoc SL(N);
17457 SDValue LHS = N->getOperand(0);
17458 SDValue RHS = N->getOperand(1);
17459
17460 if (VT.isVector())
17461 return SDValue();
17462
17463 // S_MUL_HI_[IU]32 was added in gfx9, which allows us to keep the overall
17464 // result in scalar registers for uniform values.
17465 if (!N->isDivergent() && Subtarget->hasSMulHi())
17466 return SDValue();
17467
17468 unsigned NumBits = VT.getScalarSizeInBits();
17469 if (NumBits <= 32 || NumBits > 64)
17470 return SDValue();
17471
17472 if (LHS.getOpcode() != ISD::MUL) {
17473 assert(RHS.getOpcode() == ISD::MUL);
17474 std::swap(LHS, RHS);
17475 }
17476
17477 // Avoid the fold if it would unduly increase the number of multiplies due to
17478 // multiple uses, except on hardware with full-rate multiply-add (which is
17479 // part of full-rate 64-bit ops).
17480 if (!Subtarget->hasFullRate64Ops()) {
17481 unsigned NumUsers = 0;
17482 for (SDNode *User : LHS->users()) {
17483 // There is a use that does not feed into addition, so the multiply can't
17484 // be removed. We prefer MUL + ADD + ADDC over MAD + MUL.
17485 if (!User->isAnyAdd())
17486 return SDValue();
17487
17488 // We prefer 2xMAD over MUL + 2xADD + 2xADDC (code density), and prefer
17489 // MUL + 3xADD + 3xADDC over 3xMAD.
17490 ++NumUsers;
17491 if (NumUsers >= 3)
17492 return SDValue();
17493 }
17494 }
17495
17496 SDValue MulLHS = LHS.getOperand(0);
17497 SDValue MulRHS = LHS.getOperand(1);
17498 SDValue AddRHS = RHS;
17499
17500 if (SDValue FoldedMAD = tryFoldMADwithSRL(DAG, SL, MulLHS, MulRHS, AddRHS))
17501 return FoldedMAD;
17502
17503 // Always check whether operands are small unsigned values, since that
17504 // knowledge is useful in more cases. Check for small signed values only if
17505 // doing so can unlock a shorter code sequence.
17506 bool MulLHSUnsigned32 = numBitsUnsigned(MulLHS, DAG) <= 32;
17507 bool MulRHSUnsigned32 = numBitsUnsigned(MulRHS, DAG) <= 32;
17508
17509 bool MulSignedLo = false;
17510 if (!MulLHSUnsigned32 || !MulRHSUnsigned32) {
17511 MulSignedLo =
17512 numBitsSigned(MulLHS, DAG) <= 32 && numBitsSigned(MulRHS, DAG) <= 32;
17513 }
17514
17515 // The operands and final result all have the same number of bits. If
17516 // operands need to be extended, they can be extended with garbage. The
17517 // resulting garbage in the high bits of the mad_[iu]64_[iu]32 result is
17518 // truncated away in the end.
17519 if (VT != MVT::i64) {
17520 MulLHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i64, MulLHS);
17521 MulRHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i64, MulRHS);
17522 AddRHS = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i64, AddRHS);
17523 }
17524
17525 // The basic code generated is conceptually straightforward. Pseudo code:
17526 //
17527 // accum = mad_64_32 lhs.lo, rhs.lo, accum
17528 // accum.hi = add (mul lhs.hi, rhs.lo), accum.hi
17529 // accum.hi = add (mul lhs.lo, rhs.hi), accum.hi
17530 //
17531 // The second and third lines are optional, depending on whether the factors
17532 // are {sign,zero}-extended or not.
17533 //
17534 // The actual DAG is noisier than the pseudo code, but only due to
17535 // instructions that disassemble values into low and high parts, and
17536 // assemble the final result.
17537 SDValue One = DAG.getConstant(1, SL, MVT::i32);
17538
17539 auto MulLHSLo = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, MulLHS);
17540 auto MulRHSLo = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, MulRHS);
17541 SDValue Accum =
17542 getMad64_32(DAG, SL, MVT::i64, MulLHSLo, MulRHSLo, AddRHS, MulSignedLo);
17543
17544 if (!MulSignedLo && (!MulLHSUnsigned32 || !MulRHSUnsigned32)) {
17545 auto [AccumLo, AccumHi] = DAG.SplitScalar(Accum, SL, MVT::i32, MVT::i32);
17546
17547 if (!MulLHSUnsigned32) {
17548 auto MulLHSHi =
17549 DAG.getNode(ISD::EXTRACT_ELEMENT, SL, MVT::i32, MulLHS, One);
17550 SDValue MulHi = DAG.getNode(ISD::MUL, SL, MVT::i32, MulLHSHi, MulRHSLo);
17551 AccumHi = DAG.getNode(ISD::ADD, SL, MVT::i32, MulHi, AccumHi);
17552 }
17553
17554 if (!MulRHSUnsigned32) {
17555 auto MulRHSHi =
17556 DAG.getNode(ISD::EXTRACT_ELEMENT, SL, MVT::i32, MulRHS, One);
17557 SDValue MulHi = DAG.getNode(ISD::MUL, SL, MVT::i32, MulLHSLo, MulRHSHi);
17558 AccumHi = DAG.getNode(ISD::ADD, SL, MVT::i32, MulHi, AccumHi);
17559 }
17560
17561 Accum = DAG.getBuildVector(MVT::v2i32, SL, {AccumLo, AccumHi});
17562 Accum = DAG.getBitcast(MVT::i64, Accum);
17563 }
17564
17565 if (VT != MVT::i64)
17566 Accum = DAG.getNode(ISD::TRUNCATE, SL, VT, Accum);
17567 return Accum;
17568}
17569
17570SDValue
17571SITargetLowering::foldAddSub64WithZeroLowBitsTo32(SDNode *N,
17572 DAGCombinerInfo &DCI) const {
17573 SDValue RHS = N->getOperand(1);
17574 auto *CRHS = dyn_cast<ConstantSDNode>(RHS);
17575 if (!CRHS)
17576 return SDValue();
17577
17578 // TODO: Worth using computeKnownBits? Maybe expensive since it's so
17579 // common.
17580 uint64_t Val = CRHS->getZExtValue();
17581 if (countr_zero(Val) >= 32) {
17582 SelectionDAG &DAG = DCI.DAG;
17583 SDLoc SL(N);
17584 SDValue LHS = N->getOperand(0);
17585
17586 // Avoid carry machinery if we know the low half of the add does not
17587 // contribute to the final result.
17588 //
17589 // add i64:x, K if computeTrailingZeros(K) >= 32
17590 // => build_pair (add x.hi, K.hi), x.lo
17591
17592 // Breaking the 64-bit add here with this strange constant is unlikely
17593 // to interfere with addressing mode patterns.
17594
17595 SDValue Hi = getHiHalf64(LHS, DAG);
17596 SDValue ConstHi32 = DAG.getConstant(Hi_32(Val), SL, MVT::i32);
17597 unsigned Opcode = N->getOpcode();
17598 if (Opcode == ISD::PTRADD)
17599 Opcode = ISD::ADD;
17600 SDValue AddHi =
17601 DAG.getNode(Opcode, SL, MVT::i32, Hi, ConstHi32, N->getFlags());
17602
17603 SDValue Lo = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, LHS);
17604 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i64, Lo, AddHi);
17605 }
17606
17607 return SDValue();
17608}
17609
17610// Collect the ultimate src of each of the mul node's operands, and confirm
17611// each operand is 8 bytes.
17612static std::optional<ByteProvider<SDValue>>
17613handleMulOperand(const SDValue &MulOperand) {
17614 auto Byte0 = calculateByteProvider(MulOperand, 0, 0);
17615 if (!Byte0 || Byte0->isConstantZero()) {
17616 return std::nullopt;
17617 }
17618 auto Byte1 = calculateByteProvider(MulOperand, 1, 0);
17619 if (Byte1 && !Byte1->isConstantZero()) {
17620 return std::nullopt;
17621 }
17622 return Byte0;
17623}
17624
17625static unsigned addPermMasks(unsigned First, unsigned Second) {
17626 unsigned FirstCs = First & 0x0c0c0c0c;
17627 unsigned SecondCs = Second & 0x0c0c0c0c;
17628 unsigned FirstNoCs = First & ~0x0c0c0c0c;
17629 unsigned SecondNoCs = Second & ~0x0c0c0c0c;
17630
17631 assert((FirstCs & 0xFF) | (SecondCs & 0xFF));
17632 assert((FirstCs & 0xFF00) | (SecondCs & 0xFF00));
17633 assert((FirstCs & 0xFF0000) | (SecondCs & 0xFF0000));
17634 assert((FirstCs & 0xFF000000) | (SecondCs & 0xFF000000));
17635
17636 return (FirstNoCs | SecondNoCs) | (FirstCs & SecondCs);
17637}
17638
17639struct DotSrc {
17641 int64_t PermMask;
17643};
17644
17648 SmallVectorImpl<DotSrc> &Src1s, int Step) {
17649
17650 assert(Src0.Src.has_value() && Src1.Src.has_value());
17651 // Src0s and Src1s are empty, just place arbitrarily.
17652 if (Step == 0) {
17653 Src0s.push_back({*Src0.Src, ((Src0.SrcOffset % 4) << 24) + 0x0c0c0c,
17654 Src0.SrcOffset / 4});
17655 Src1s.push_back({*Src1.Src, ((Src1.SrcOffset % 4) << 24) + 0x0c0c0c,
17656 Src1.SrcOffset / 4});
17657 return;
17658 }
17659
17660 for (int BPI = 0; BPI < 2; BPI++) {
17661 std::pair<ByteProvider<SDValue>, ByteProvider<SDValue>> BPP = {Src0, Src1};
17662 if (BPI == 1) {
17663 BPP = {Src1, Src0};
17664 }
17665 unsigned ZeroMask = 0x0c0c0c0c;
17666 unsigned FMask = 0xFF << (8 * (3 - Step));
17667
17668 unsigned FirstMask =
17669 (BPP.first.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask);
17670 unsigned SecondMask =
17671 (BPP.second.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask);
17672 // Attempt to find Src vector which contains our SDValue, if so, add our
17673 // perm mask to the existing one. If we are unable to find a match for the
17674 // first SDValue, attempt to find match for the second.
17675 int FirstGroup = -1;
17676 for (int I = 0; I < 2; I++) {
17677 SmallVectorImpl<DotSrc> &Srcs = I == 0 ? Src0s : Src1s;
17678 auto MatchesFirst = [&BPP](DotSrc &IterElt) {
17679 return IterElt.SrcOp == *BPP.first.Src &&
17680 (IterElt.DWordOffset == (BPP.first.SrcOffset / 4));
17681 };
17682
17683 auto *Match = llvm::find_if(Srcs, MatchesFirst);
17684 if (Match != Srcs.end()) {
17685 Match->PermMask = addPermMasks(FirstMask, Match->PermMask);
17686 FirstGroup = I;
17687 break;
17688 }
17689 }
17690 if (FirstGroup != -1) {
17691 SmallVectorImpl<DotSrc> &Srcs = FirstGroup == 1 ? Src0s : Src1s;
17692 auto MatchesSecond = [&BPP](DotSrc &IterElt) {
17693 return IterElt.SrcOp == *BPP.second.Src &&
17694 (IterElt.DWordOffset == (BPP.second.SrcOffset / 4));
17695 };
17696 auto *Match = llvm::find_if(Srcs, MatchesSecond);
17697 if (Match != Srcs.end()) {
17698 Match->PermMask = addPermMasks(SecondMask, Match->PermMask);
17699 } else
17700 Srcs.push_back({*BPP.second.Src, SecondMask, BPP.second.SrcOffset / 4});
17701 return;
17702 }
17703 }
17704
17705 // If we have made it here, then we could not find a match in Src0s or Src1s
17706 // for either Src0 or Src1, so just place them arbitrarily.
17707
17708 unsigned ZeroMask = 0x0c0c0c0c;
17709 unsigned FMask = 0xFF << (8 * (3 - Step));
17710
17711 Src0s.push_back(
17712 {*Src0.Src,
17713 ((Src0.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask)),
17714 Src0.SrcOffset / 4});
17715 Src1s.push_back(
17716 {*Src1.Src,
17717 ((Src1.SrcOffset % 4) << (8 * (3 - Step)) | (ZeroMask & ~FMask)),
17718 Src1.SrcOffset / 4});
17719}
17720
17722 SmallVectorImpl<DotSrc> &Srcs, bool IsSigned,
17723 bool IsAny) {
17724
17725 // If we just have one source, just permute it accordingly.
17726 if (Srcs.size() == 1) {
17727 auto *Elt = Srcs.begin();
17728 auto EltOp = getDWordFromOffset(DAG, SL, Elt->SrcOp, Elt->DWordOffset);
17729
17730 // v_perm will produce the original value
17731 if (Elt->PermMask == 0x3020100)
17732 return EltOp;
17733
17734 return DAG.getNode(AMDGPUISD::PERM, SL, MVT::i32, EltOp, EltOp,
17735 DAG.getConstant(Elt->PermMask, SL, MVT::i32));
17736 }
17737
17738 auto *FirstElt = Srcs.begin();
17739 auto *SecondElt = std::next(FirstElt);
17740
17742
17743 // If we have multiple sources in the chain, combine them via perms (using
17744 // calculated perm mask) and Ors.
17745 while (true) {
17746 auto FirstMask = FirstElt->PermMask;
17747 auto SecondMask = SecondElt->PermMask;
17748
17749 unsigned FirstCs = FirstMask & 0x0c0c0c0c;
17750 unsigned FirstPlusFour = FirstMask | 0x04040404;
17751 // 0x0c + 0x04 = 0x10, so anding with 0x0F will produced 0x00 for any
17752 // original 0x0C.
17753 FirstMask = (FirstPlusFour & 0x0F0F0F0F) | FirstCs;
17754
17755 auto PermMask = addPermMasks(FirstMask, SecondMask);
17756 auto FirstVal =
17757 getDWordFromOffset(DAG, SL, FirstElt->SrcOp, FirstElt->DWordOffset);
17758 auto SecondVal =
17759 getDWordFromOffset(DAG, SL, SecondElt->SrcOp, SecondElt->DWordOffset);
17760
17761 Perms.push_back(DAG.getNode(AMDGPUISD::PERM, SL, MVT::i32, FirstVal,
17762 SecondVal,
17763 DAG.getConstant(PermMask, SL, MVT::i32)));
17764
17765 FirstElt = std::next(SecondElt);
17766 if (FirstElt == Srcs.end())
17767 break;
17768
17769 SecondElt = std::next(FirstElt);
17770 // If we only have a FirstElt, then just combine that into the cumulative
17771 // source node.
17772 if (SecondElt == Srcs.end()) {
17773 auto EltOp =
17774 getDWordFromOffset(DAG, SL, FirstElt->SrcOp, FirstElt->DWordOffset);
17775
17776 Perms.push_back(
17777 DAG.getNode(AMDGPUISD::PERM, SL, MVT::i32, EltOp, EltOp,
17778 DAG.getConstant(FirstElt->PermMask, SL, MVT::i32)));
17779 break;
17780 }
17781 }
17782
17783 assert(Perms.size() == 1 || Perms.size() == 2);
17784 return Perms.size() == 2
17785 ? DAG.getNode(ISD::OR, SL, MVT::i32, Perms[0], Perms[1])
17786 : Perms[0];
17787}
17788
17789static void fixMasks(SmallVectorImpl<DotSrc> &Srcs, unsigned ChainLength) {
17790 for (auto &[EntryVal, EntryMask, EntryOffset] : Srcs) {
17791 EntryMask = EntryMask >> ((4 - ChainLength) * 8);
17792 auto ZeroMask = ChainLength == 2 ? 0x0c0c0000 : 0x0c000000;
17793 EntryMask += ZeroMask;
17794 }
17795}
17796
17797static bool isMul(const SDValue Op) {
17798 auto Opcode = Op.getOpcode();
17799
17800 return (Opcode == ISD::MUL || Opcode == AMDGPUISD::MUL_U24 ||
17801 Opcode == AMDGPUISD::MUL_I24);
17802}
17803
17804static std::optional<bool>
17806 ByteProvider<SDValue> &Src1, const SDValue &S0Op,
17807 const SDValue &S1Op, const SelectionDAG &DAG) {
17808 // If we both ops are i8s (pre legalize-dag), then the signedness semantics
17809 // of the dot4 is irrelevant.
17810 if (S0Op.getValueSizeInBits() == 8 && S1Op.getValueSizeInBits() == 8)
17811 return false;
17812
17813 auto Known0 = DAG.computeKnownBits(S0Op, 0);
17814 bool S0IsUnsigned = Known0.countMinLeadingZeros() > 0;
17815 bool S0IsSigned = Known0.countMinLeadingOnes() > 0;
17816 auto Known1 = DAG.computeKnownBits(S1Op, 0);
17817 bool S1IsUnsigned = Known1.countMinLeadingZeros() > 0;
17818 bool S1IsSigned = Known1.countMinLeadingOnes() > 0;
17819
17820 assert(!(S0IsUnsigned && S0IsSigned));
17821 assert(!(S1IsUnsigned && S1IsSigned));
17822
17823 // There are 9 possible permutations of
17824 // {S0IsUnsigned, S0IsSigned, S1IsUnsigned, S1IsSigned}
17825
17826 // In two permutations, the sign bits are known to be the same for both Ops,
17827 // so simply return Signed / Unsigned corresponding to the MSB
17828
17829 if ((S0IsUnsigned && S1IsUnsigned) || (S0IsSigned && S1IsSigned))
17830 return S0IsSigned;
17831
17832 // In another two permutations, the sign bits are known to be opposite. In
17833 // this case return std::nullopt to indicate a bad match.
17834
17835 if ((S0IsUnsigned && S1IsSigned) || (S0IsSigned && S1IsUnsigned))
17836 return std::nullopt;
17837
17838 // In the remaining five permutations, we don't know the value of the sign
17839 // bit for at least one Op. Since we have a valid ByteProvider, we know that
17840 // the upper bits must be extension bits. Thus, the only ways for the sign
17841 // bit to be unknown is if it was sign extended from unknown value, or if it
17842 // was any extended. In either case, it is correct to use the signed
17843 // version of the signedness semantics of dot4
17844
17845 // In two of such permutations, we known the sign bit is set for
17846 // one op, and the other is unknown. It is okay to used signed version of
17847 // dot4.
17848 if ((S0IsSigned && !(S1IsSigned || S1IsUnsigned)) ||
17849 ((S1IsSigned && !(S0IsSigned || S0IsUnsigned))))
17850 return true;
17851
17852 // In one such permutation, we don't know either of the sign bits. It is okay
17853 // to used the signed version of dot4.
17854 if ((!(S1IsSigned || S1IsUnsigned) && !(S0IsSigned || S0IsUnsigned)))
17855 return true;
17856
17857 // In two of such permutations, we known the sign bit is unset for
17858 // one op, and the other is unknown. Return std::nullopt to indicate a
17859 // bad match.
17860 if ((S0IsUnsigned && !(S1IsSigned || S1IsUnsigned)) ||
17861 ((S1IsUnsigned && !(S0IsSigned || S0IsUnsigned))))
17862 return std::nullopt;
17863
17864 llvm_unreachable("Fully covered condition");
17865}
17866
17867SDValue SITargetLowering::performAddCombine(SDNode *N,
17868 DAGCombinerInfo &DCI) const {
17869 SelectionDAG &DAG = DCI.DAG;
17870 EVT VT = N->getValueType(0);
17871 SDLoc SL(N);
17872 SDValue LHS = N->getOperand(0);
17873 SDValue RHS = N->getOperand(1);
17874
17875 if (LHS.getOpcode() == ISD::MUL || RHS.getOpcode() == ISD::MUL) {
17876 if (Subtarget->hasMad64_32()) {
17877 if (SDValue Folded = tryFoldToMad64_32(N, DCI))
17878 return Folded;
17879 }
17880 }
17881
17882 if (SDValue V = reassociateScalarOps(N, DAG)) {
17883 return V;
17884 }
17885
17886 if (VT == MVT::i64) {
17887 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(N, DCI))
17888 return Folded;
17889 }
17890
17891 // dot4 produces a 32-bit result, so a wider VT can't be folded.
17892 if (!VT.isVector() && VT.getSizeInBits() <= 32 &&
17893 (isMul(LHS) || isMul(RHS)) && Subtarget->hasDot7Insts() &&
17894 (Subtarget->hasDot1Insts() || Subtarget->hasDot8Insts())) {
17895 SDValue TempNode(N, 0);
17896 std::optional<bool> IsSigned;
17900
17901 // Match the v_dot4 tree, while collecting src nodes.
17902 int ChainLength = 0;
17903 for (int I = 0; I < 4; I++) {
17904 auto MulIdx = isMul(LHS) ? 0 : isMul(RHS) ? 1 : -1;
17905 if (MulIdx == -1)
17906 break;
17907 auto Src0 = handleMulOperand(TempNode->getOperand(MulIdx)->getOperand(0));
17908 if (!Src0)
17909 break;
17910 auto Src1 = handleMulOperand(TempNode->getOperand(MulIdx)->getOperand(1));
17911 if (!Src1)
17912 break;
17913
17914 auto IterIsSigned = checkDot4MulSignedness(
17915 TempNode->getOperand(MulIdx), *Src0, *Src1,
17916 TempNode->getOperand(MulIdx)->getOperand(0),
17917 TempNode->getOperand(MulIdx)->getOperand(1), DAG);
17918 if (!IterIsSigned)
17919 break;
17920 if (!IsSigned)
17921 IsSigned = *IterIsSigned;
17922 if (*IterIsSigned != *IsSigned)
17923 break;
17924 placeSources(*Src0, *Src1, Src0s, Src1s, I);
17925 auto AddIdx = 1 - MulIdx;
17926 // Allow the special case where add (add (mul24, 0), mul24) became ->
17927 // add (mul24, mul24).
17928 if (I == 2 && isMul(TempNode->getOperand(AddIdx))) {
17929 Src2s.push_back(TempNode->getOperand(AddIdx));
17930 auto Src0 =
17931 handleMulOperand(TempNode->getOperand(AddIdx)->getOperand(0));
17932 if (!Src0)
17933 break;
17934 auto Src1 =
17935 handleMulOperand(TempNode->getOperand(AddIdx)->getOperand(1));
17936 if (!Src1)
17937 break;
17938 auto IterIsSigned = checkDot4MulSignedness(
17939 TempNode->getOperand(AddIdx), *Src0, *Src1,
17940 TempNode->getOperand(AddIdx)->getOperand(0),
17941 TempNode->getOperand(AddIdx)->getOperand(1), DAG);
17942 if (!IterIsSigned)
17943 break;
17944 assert(IsSigned);
17945 if (*IterIsSigned != *IsSigned)
17946 break;
17947 placeSources(*Src0, *Src1, Src0s, Src1s, I + 1);
17948 Src2s.push_back(DAG.getConstant(0, SL, MVT::i32));
17949 ChainLength = I + 2;
17950 break;
17951 }
17952
17953 TempNode = TempNode->getOperand(AddIdx);
17954 Src2s.push_back(TempNode);
17955 ChainLength = I + 1;
17956 // The loop body treats TempNode's operands as addends.
17957 if (TempNode.getOpcode() != ISD::ADD)
17958 break;
17959 LHS = TempNode->getOperand(0);
17960 RHS = TempNode->getOperand(1);
17961 }
17962
17963 if (ChainLength < 2)
17964 return SDValue();
17965
17966 // Masks were constructed with assumption that we would find a chain of
17967 // length 4. If not, then we need to 0 out the MSB bits (via perm mask of
17968 // 0x0c) so they do not affect dot calculation.
17969 if (ChainLength < 4) {
17970 fixMasks(Src0s, ChainLength);
17971 fixMasks(Src1s, ChainLength);
17972 }
17973
17974 SDValue Src0, Src1;
17975
17976 // If we are just using a single source for both, and have permuted the
17977 // bytes consistently, we can just use the sources without permuting
17978 // (commutation).
17979 bool UseOriginalSrc = false;
17980 if (ChainLength == 4 && Src0s.size() == 1 && Src1s.size() == 1 &&
17981 Src0s.begin()->PermMask == Src1s.begin()->PermMask &&
17982 Src0s.begin()->SrcOp.getValueSizeInBits() >= 32 &&
17983 Src1s.begin()->SrcOp.getValueSizeInBits() >= 32) {
17984 SmallVector<unsigned, 4> SrcBytes;
17985 auto Src0Mask = Src0s.begin()->PermMask;
17986 SrcBytes.push_back(Src0Mask & 0xFF000000);
17987 bool UniqueEntries = true;
17988 for (auto I = 1; I < 4; I++) {
17989 auto NextByte = Src0Mask & (0xFF << ((3 - I) * 8));
17990
17991 if (is_contained(SrcBytes, NextByte)) {
17992 UniqueEntries = false;
17993 break;
17994 }
17995 SrcBytes.push_back(NextByte);
17996 }
17997
17998 if (UniqueEntries) {
17999 UseOriginalSrc = true;
18000
18001 auto *FirstElt = Src0s.begin();
18002 auto FirstEltOp =
18003 getDWordFromOffset(DAG, SL, FirstElt->SrcOp, FirstElt->DWordOffset);
18004
18005 auto *SecondElt = Src1s.begin();
18006 auto SecondEltOp = getDWordFromOffset(DAG, SL, SecondElt->SrcOp,
18007 SecondElt->DWordOffset);
18008
18009 Src0 = DAG.getBitcastedAnyExtOrTrunc(FirstEltOp, SL,
18010 MVT::getIntegerVT(32));
18011 Src1 = DAG.getBitcastedAnyExtOrTrunc(SecondEltOp, SL,
18012 MVT::getIntegerVT(32));
18013 }
18014 }
18015
18016 if (!UseOriginalSrc) {
18017 Src0 = resolveSources(DAG, SL, Src0s, false, true);
18018 Src1 = resolveSources(DAG, SL, Src1s, false, true);
18019 }
18020
18021 assert(IsSigned);
18022 SDValue Src2 =
18023 DAG.getExtOrTrunc(*IsSigned, Src2s[ChainLength - 1], SL, MVT::i32);
18024
18025 SDValue IID = DAG.getTargetConstant(*IsSigned ? Intrinsic::amdgcn_sdot4
18026 : Intrinsic::amdgcn_udot4,
18027 SL, MVT::i64);
18028
18029 assert(!VT.isVector());
18030 auto Dot = DAG.getNode(ISD::INTRINSIC_WO_CHAIN, SL, MVT::i32, IID, Src0,
18031 Src1, Src2, DAG.getTargetConstant(0, SL, MVT::i1));
18032
18033 return DAG.getExtOrTrunc(*IsSigned, Dot, SL, VT);
18034 }
18035
18036 if (VT != MVT::i32 || !DCI.isAfterLegalizeDAG())
18037 return SDValue();
18038
18039 // add x, zext (setcc) => uaddo_carry x, 0, setcc
18040 // add x, sext (setcc) => usubo_carry x, 0, setcc
18041 unsigned Opc = LHS.getOpcode();
18044 std::swap(RHS, LHS);
18045
18046 Opc = RHS.getOpcode();
18047 switch (Opc) {
18048 default:
18049 break;
18050 case ISD::ZERO_EXTEND:
18051 case ISD::SIGN_EXTEND:
18052 case ISD::ANY_EXTEND: {
18053 auto Cond = RHS.getOperand(0);
18054 // If this won't be a real VOPC output, we would still need to insert an
18055 // extra instruction anyway.
18056 if (!isBoolSGPR(Cond))
18057 break;
18058 SDVTList VTList = DAG.getVTList(MVT::i32, MVT::i1);
18059 SDValue Args[] = {LHS, DAG.getConstant(0, SL, MVT::i32), Cond};
18061 return DAG.getNode(Opc, SL, VTList, Args);
18062 }
18063 case ISD::UADDO_CARRY: {
18064 // add x, (uaddo_carry y, 0, cc) => uaddo_carry x, y, cc
18065 if (!isNullConstant(RHS.getOperand(1)))
18066 break;
18067 SDValue Args[] = {LHS, RHS.getOperand(0), RHS.getOperand(2)};
18068 return DAG.getNode(ISD::UADDO_CARRY, SDLoc(N), RHS->getVTList(), Args);
18069 }
18070 }
18071 return SDValue();
18072}
18073
18074SDValue SITargetLowering::performPtrAddCombine(SDNode *N,
18075 DAGCombinerInfo &DCI) const {
18076 SelectionDAG &DAG = DCI.DAG;
18077 SDLoc DL(N);
18078 EVT VT = N->getValueType(0);
18079 SDValue N0 = N->getOperand(0);
18080 SDValue N1 = N->getOperand(1);
18081
18082 // The following folds transform PTRADDs into regular arithmetic in cases
18083 // where the PTRADD wouldn't be folded as an immediate offset into memory
18084 // instructions anyway. They are target-specific in that other targets might
18085 // prefer to not lose information about the pointer arithmetic.
18086
18087 // Fold (ptradd x, shl(0 - v, k)) -> sub(x, shl(v, k)).
18088 // Adapted from DAGCombiner::visitADDLikeCommutative.
18089 SDValue V, K;
18090 if (sd_match(N1, m_Shl(m_Neg(m_Value(V)), m_Value(K)))) {
18091 SDNodeFlags ShlFlags = N1->getFlags();
18092 // If the original shl is NUW and NSW, the first k+1 bits of 0-v are all 0,
18093 // so v is either 0 or the first k+1 bits of v are all 1 -> NSW can be
18094 // preserved.
18095 SDNodeFlags NewShlFlags =
18096 ShlFlags.hasNoUnsignedWrap() && ShlFlags.hasNoSignedWrap()
18098 : SDNodeFlags();
18099 SDValue Inner = DAG.getNode(ISD::SHL, DL, VT, V, K, NewShlFlags);
18100 DCI.AddToWorklist(Inner.getNode());
18101 return DAG.getNode(ISD::SUB, DL, VT, N0, Inner);
18102 }
18103
18104 // Fold into Mad64 if the right-hand side is a MUL. Analogous to a fold in
18105 // performAddCombine.
18106 if (N1.getOpcode() == ISD::MUL) {
18107 if (Subtarget->hasMad64_32()) {
18108 if (SDValue Folded = tryFoldToMad64_32(N, DCI))
18109 return Folded;
18110 }
18111 }
18112
18113 // If the 32 low bits of the constant are all zero, there is nothing to fold
18114 // into an immediate offset, so it's better to eliminate the unnecessary
18115 // addition for the lower 32 bits than to preserve the PTRADD.
18116 // Analogous to a fold in performAddCombine.
18117 if (VT == MVT::i64) {
18118 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(N, DCI))
18119 return Folded;
18120 }
18121
18122 if (N1.getOpcode() != ISD::ADD || !N1.hasOneUse())
18123 return SDValue();
18124
18125 SDValue X = N0;
18126 SDValue Y = N1.getOperand(0);
18127 SDValue Z = N1.getOperand(1);
18128 bool YIsConstant = DAG.isConstantIntBuildVectorOrConstantInt(Y);
18129 bool ZIsConstant = DAG.isConstantIntBuildVectorOrConstantInt(Z);
18130
18131 if (!YIsConstant && !ZIsConstant && !X->isDivergent() &&
18132 Y->isDivergent() != Z->isDivergent()) {
18133 // Reassociate (ptradd x, (add y, z)) -> (ptradd (ptradd x, y), z) if x and
18134 // y are uniform and z isn't.
18135 // Reassociate (ptradd x, (add y, z)) -> (ptradd (ptradd x, z), y) if x and
18136 // z are uniform and y isn't.
18137 // The goal is to push uniform operands up in the computation, so that they
18138 // can be handled with scalar operations. We can't use reassociateScalarOps
18139 // for this since it requires two identical commutative operations to
18140 // reassociate.
18141 if (Y->isDivergent())
18142 std::swap(Y, Z);
18143 // If both additions in the original were NUW, reassociation preserves that.
18144 SDNodeFlags ReassocFlags =
18145 (N->getFlags() & N1->getFlags()) & SDNodeFlags::NoUnsignedWrap;
18146 SDValue UniformInner = DAG.getMemBasePlusOffset(X, Y, DL, ReassocFlags);
18147 DCI.AddToWorklist(UniformInner.getNode());
18148 return DAG.getMemBasePlusOffset(UniformInner, Z, DL, ReassocFlags);
18149 }
18150
18151 return SDValue();
18152}
18153
18154static bool isCtlzOpc(unsigned Opc) {
18155 return Opc == ISD::CTLZ || Opc == ISD::CTLZ_ZERO_POISON;
18156}
18157
18158SDValue SITargetLowering::performSubCombine(SDNode *N,
18159 DAGCombinerInfo &DCI) const {
18160 SelectionDAG &DAG = DCI.DAG;
18161 EVT VT = N->getValueType(0);
18162
18163 if (VT == MVT::i64) {
18164 if (SDValue Folded = foldAddSub64WithZeroLowBitsTo32(N, DCI))
18165 return Folded;
18166 }
18167
18168 if (VT != MVT::i32)
18169 return SDValue();
18170
18171 SDLoc SL(N);
18172 SDValue LHS = N->getOperand(0);
18173 SDValue RHS = N->getOperand(1);
18174
18175 // sub x, zext (setcc) => usubo_carry x, 0, setcc
18176 // sub x, sext (setcc) => uaddo_carry x, 0, setcc
18177 unsigned Opc = RHS.getOpcode();
18178 switch (Opc) {
18179 default:
18180 break;
18181 case ISD::ZERO_EXTEND:
18182 case ISD::SIGN_EXTEND:
18183 case ISD::ANY_EXTEND: {
18184 auto Cond = RHS.getOperand(0);
18185 // If this won't be a real VOPC output, we would still need to insert an
18186 // extra instruction anyway.
18187 if (!isBoolSGPR(Cond))
18188 break;
18189 SDVTList VTList = DAG.getVTList(MVT::i32, MVT::i1);
18190 SDValue Args[] = {LHS, DAG.getConstant(0, SL, MVT::i32), Cond};
18192 return DAG.getNode(Opc, SL, VTList, Args);
18193 }
18194 }
18195
18196 if (LHS.getOpcode() == ISD::USUBO_CARRY) {
18197 // sub (usubo_carry x, 0, cc), y => usubo_carry x, y, cc
18198 if (!isNullConstant(LHS.getOperand(1)))
18199 return SDValue();
18200 SDValue Args[] = {LHS.getOperand(0), RHS, LHS.getOperand(2)};
18201 return DAG.getNode(ISD::USUBO_CARRY, SDLoc(N), LHS->getVTList(), Args);
18202 }
18203
18204 // sub (ctlz (xor x, (sra x, 31))), 1 -> ctls x.
18205 if (isOneConstant(RHS) && isCtlzOpc(LHS.getOpcode())) {
18206 SDValue CtlzSrc = LHS.getOperand(0);
18207 // Check for xor x, (sra x, 31) pattern.
18208 if (CtlzSrc.getOpcode() == ISD::XOR) {
18209 SDValue X = CtlzSrc.getOperand(0);
18210 SDValue SignExt = CtlzSrc.getOperand(1);
18211 // Try both ordering of XOR operands.
18212 if (SignExt.getOpcode() != ISD::SRA)
18213 std::swap(X, SignExt);
18214 if (SignExt.getOpcode() == ISD::SRA && SignExt.getOperand(0) == X) {
18215 ConstantSDNode *ShiftAmt =
18217 unsigned BitWidth = X.getValueType().getScalarSizeInBits();
18218 if (ShiftAmt && ShiftAmt->getZExtValue() == BitWidth - 1)
18219 return DAG.getNode(ISD::CTLS, SL, VT, X);
18220 }
18221 }
18222 }
18223
18224 return SDValue();
18225}
18226
18227SDValue SITargetLowering::performFAddCombine(SDNode *N,
18228 DAGCombinerInfo &DCI) const {
18229 if (DCI.getDAGCombineLevel() < AfterLegalizeDAG)
18230 return SDValue();
18231
18232 SelectionDAG &DAG = DCI.DAG;
18233 EVT VT = N->getValueType(0);
18234
18235 SDLoc SL(N);
18236 SDValue LHS = N->getOperand(0);
18237 SDValue RHS = N->getOperand(1);
18238
18239 // These should really be instruction patterns, but writing patterns with
18240 // source modifiers is a pain.
18241
18242 // fadd (fadd (a, a), b) -> mad 2.0, a, b
18243 if (LHS.getOpcode() == ISD::FADD) {
18244 SDValue A = LHS.getOperand(0);
18245 if (A == LHS.getOperand(1)) {
18246 unsigned FusedOp = getFusedOpcode(DAG, N, LHS.getNode());
18247 if (FusedOp != 0) {
18248 const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
18249 return DAG.getNode(FusedOp, SL, VT, A, Two, RHS);
18250 }
18251 }
18252 }
18253
18254 // fadd (b, fadd (a, a)) -> mad 2.0, a, b
18255 if (RHS.getOpcode() == ISD::FADD) {
18256 SDValue A = RHS.getOperand(0);
18257 if (A == RHS.getOperand(1)) {
18258 unsigned FusedOp = getFusedOpcode(DAG, N, RHS.getNode());
18259 if (FusedOp != 0) {
18260 const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
18261 return DAG.getNode(FusedOp, SL, VT, A, Two, LHS);
18262 }
18263 }
18264 }
18265
18266 return SDValue();
18267}
18268
18269SDValue SITargetLowering::performFSubCombine(SDNode *N,
18270 DAGCombinerInfo &DCI) const {
18271 if (DCI.getDAGCombineLevel() < AfterLegalizeDAG)
18272 return SDValue();
18273
18274 SelectionDAG &DAG = DCI.DAG;
18275 SDLoc SL(N);
18276 EVT VT = N->getValueType(0);
18277 assert(!VT.isVector());
18278
18279 // Try to get the fneg to fold into the source modifier. This undoes generic
18280 // DAG combines and folds them into the mad.
18281 //
18282 // Only do this if we are not trying to support denormals. v_mad_f32 does
18283 // not support denormals ever.
18284 SDValue LHS = N->getOperand(0);
18285 SDValue RHS = N->getOperand(1);
18286 if (LHS.getOpcode() == ISD::FADD) {
18287 // (fsub (fadd a, a), c) -> mad 2.0, a, (fneg c)
18288 SDValue A = LHS.getOperand(0);
18289 if (A == LHS.getOperand(1)) {
18290 unsigned FusedOp = getFusedOpcode(DAG, N, LHS.getNode());
18291 if (FusedOp != 0) {
18292 const SDValue Two = DAG.getConstantFP(2.0, SL, VT);
18293 SDValue NegRHS = DAG.getNode(ISD::FNEG, SL, VT, RHS);
18294
18295 return DAG.getNode(FusedOp, SL, VT, A, Two, NegRHS);
18296 }
18297 }
18298 }
18299
18300 if (RHS.getOpcode() == ISD::FADD) {
18301 // (fsub c, (fadd a, a)) -> mad -2.0, a, c
18302
18303 SDValue A = RHS.getOperand(0);
18304 if (A == RHS.getOperand(1)) {
18305 unsigned FusedOp = getFusedOpcode(DAG, N, RHS.getNode());
18306 if (FusedOp != 0) {
18307 const SDValue NegTwo = DAG.getConstantFP(-2.0, SL, VT);
18308 return DAG.getNode(FusedOp, SL, VT, A, NegTwo, LHS);
18309 }
18310 }
18311 }
18312
18313 return SDValue();
18314}
18315
18316SDValue SITargetLowering::performFDivCombine(SDNode *N,
18317 DAGCombinerInfo &DCI) const {
18318 SelectionDAG &DAG = DCI.DAG;
18319 SDLoc SL(N);
18320 EVT VT = N->getValueType(0);
18321
18322 if (VT != MVT::f16 && VT != MVT::bf16)
18323 return SDValue();
18324
18325 SDValue LHS = N->getOperand(0);
18326 SDValue RHS = N->getOperand(1);
18327
18328 SDNodeFlags Flags = N->getFlags();
18329 SDNodeFlags RHSFlags = RHS->getFlags();
18330 if (!Flags.hasAllowContract() || !RHSFlags.hasAllowContract() ||
18331 !RHS->hasOneUse())
18332 return SDValue();
18333
18334 if (const ConstantFPSDNode *CLHS = dyn_cast<ConstantFPSDNode>(LHS)) {
18335 bool IsNegative = false;
18336 if (CLHS->isOne() || (IsNegative = CLHS->isMinusOne())) {
18337 // fdiv contract 1.0, (sqrt contract x) -> rsq
18338 // fdiv contract -1.0, (sqrt contract x) -> fneg(rsq)
18339 if (RHS.getOpcode() == ISD::FSQRT) {
18340 // TODO: Or in RHS flags, somehow missing from SDNodeFlags
18341 SDValue SqrtOp = RHS.getOperand(0);
18342 SDValue Rsq;
18343 if (isOperationLegal(ISD::FSQRT, VT)) {
18344 // fsqrt legality correlates to rsq availability of the same type.
18345 Rsq = DAG.getNode(AMDGPUISD::RSQ, SL, VT, SqrtOp, Flags);
18346 } else if (VT == MVT::f16) {
18347 // Targets without 16-bit instructions (gfx6/gfx7) have no f16 rsq,
18348 // but v_rsq_f32 is more than accurate enough for f16. Unlike bf16,
18349 // every f16 value (including denormals) extends to a normal f32, and
18350 // an f16 rsq result is never denormal, so the f32 reciprocal square
18351 // root needs no denormal handling. Compute it in f32 and round back.
18352 SDValue Ext =
18353 DAG.getNode(ISD::FP_EXTEND, SL, MVT::f32, SqrtOp, Flags);
18354 SDValue F32Rsq =
18355 DAG.getNode(AMDGPUISD::RSQ, SL, MVT::f32, Ext, Flags);
18356 Rsq = DAG.getNode(ISD::FP_ROUND, SL, VT, F32Rsq,
18357 DAG.getTargetConstant(0, SL, MVT::i32), Flags);
18358 } else {
18359 // bf16 shares f32's exponent range, so bf16 denormals would extend to
18360 // f32 denormals that v_rsq_f32 does not handle. Leave it expanded.
18361 return SDValue();
18362 }
18363 return IsNegative ? DAG.getNode(ISD::FNEG, SL, VT, Rsq, Flags) : Rsq;
18364 }
18365 }
18366 }
18367
18368 return SDValue();
18369}
18370
18371SDValue SITargetLowering::performFMulCombine(SDNode *N,
18372 DAGCombinerInfo &DCI) const {
18373 SelectionDAG &DAG = DCI.DAG;
18374 EVT VT = N->getValueType(0);
18375 EVT ScalarVT = VT.getScalarType();
18376 EVT IntVT = VT.changeElementType(*DAG.getContext(), MVT::i32);
18377
18378 if (!N->isDivergent() && getSubtarget()->hasSALUFloatInsts() &&
18379 (ScalarVT == MVT::f32 || ScalarVT == MVT::f16)) {
18380 // Prefer to use s_mul_f16/f32 instead of v_ldexp_f16/f32.
18381 return SDValue();
18382 }
18383
18384 SDValue LHS = N->getOperand(0);
18385 SDValue RHS = N->getOperand(1);
18386
18387 // It is cheaper to realize i32 inline constants as compared against
18388 // materializing f16 or f64 (or even non-inline f32) values,
18389 // possible via ldexp usage, as shown below :
18390 //
18391 // Given : A = 2^a & B = 2^b ; where a and b are integers.
18392 // fmul x, (select y, A, B) -> ldexp( x, (select i32 y, a, b) )
18393 // fmul x, (select y, -A, -B) -> ldexp( (fneg x), (select i32 y, a, b) )
18394 if ((ScalarVT == MVT::f64 || ScalarVT == MVT::f32 || ScalarVT == MVT::f16) &&
18395 (RHS.hasOneUse() && RHS.getOpcode() == ISD::SELECT)) {
18396 const ConstantFPSDNode *TrueNode = isConstOrConstSplatFP(RHS.getOperand(1));
18397 if (!TrueNode)
18398 return SDValue();
18399 const ConstantFPSDNode *FalseNode =
18400 isConstOrConstSplatFP(RHS.getOperand(2));
18401 if (!FalseNode)
18402 return SDValue();
18403
18404 if (TrueNode->isNegative() != FalseNode->isNegative())
18405 return SDValue();
18406
18407 // For f32, only non-inline constants should be transformed.
18408 const SIInstrInfo *TII = getSubtarget()->getInstrInfo();
18409 if (ScalarVT == MVT::f32 &&
18410 TII->isInlineConstant(TrueNode->getValueAPF()) &&
18411 TII->isInlineConstant(FalseNode->getValueAPF()))
18412 return SDValue();
18413
18414 int TrueNodeExpVal = TrueNode->getValueAPF().getExactLog2Abs();
18415 if (TrueNodeExpVal == INT_MIN)
18416 return SDValue();
18417 int FalseNodeExpVal = FalseNode->getValueAPF().getExactLog2Abs();
18418 if (FalseNodeExpVal == INT_MIN)
18419 return SDValue();
18420
18421 SDLoc SL(N);
18422 SDValue SelectNode =
18423 DAG.getNode(ISD::SELECT, SL, IntVT, RHS.getOperand(0),
18424 DAG.getSignedConstant(TrueNodeExpVal, SL, IntVT),
18425 DAG.getSignedConstant(FalseNodeExpVal, SL, IntVT));
18426
18427 LHS = TrueNode->isNegative()
18428 ? DAG.getNode(ISD::FNEG, SL, VT, LHS, LHS->getFlags())
18429 : LHS;
18430
18431 return DAG.getNode(ISD::FLDEXP, SL, VT, LHS, SelectNode, N->getFlags());
18432 }
18433
18434 return SDValue();
18435}
18436
18437SDValue SITargetLowering::performFMACombine(SDNode *N,
18438 DAGCombinerInfo &DCI) const {
18439 SelectionDAG &DAG = DCI.DAG;
18440 EVT VT = N->getValueType(0);
18441 SDLoc SL(N);
18442
18443 if (!Subtarget->hasDot10Insts() || VT != MVT::f32)
18444 return SDValue();
18445
18446 // FMA((F32)S0.x, (F32)S1. x, FMA((F32)S0.y, (F32)S1.y, (F32)z)) ->
18447 // FDOT2((V2F16)S0, (V2F16)S1, (F32)z))
18448 SDValue Op1 = N->getOperand(0);
18449 SDValue Op2 = N->getOperand(1);
18450 SDValue FMA = N->getOperand(2);
18451
18452 if (FMA.getOpcode() != ISD::FMA || Op1.getOpcode() != ISD::FP_EXTEND ||
18453 Op2.getOpcode() != ISD::FP_EXTEND)
18454 return SDValue();
18455
18456 // The fdot2 fold (fma_mix -> dot2) is only safe when both instructions agree
18457 // on how f16 subnormal inputs are handled. However, if both FMAs carry afn
18458 // the caller accepts approximate results, so any subnormal flushing
18459 // introduced by dot2 is acceptable regardless of mode.
18460 //
18461 // gfx90a (CDNA2) is the sole exception (dot2UnconditionalFlush): v_dot2c
18462 // unconditionally flushes f16 subnormal inputs to zero regardless of MODE,
18463 // while v_fma_mix_f32 preserves them when ieee=1 (the default compute kernel
18464 // mode). The fold is safe only when f32 denorm = PreserveSign, which implies
18465 // ieee=0 so both flush.
18466 //
18467 // All other GPUs: v_dot2 does NOT flush f16 subnormal inputs. v_fma_mix_f32
18468 // flushes them only when f32 denorm = PreserveSign. The fold is safe only
18469 // when f32 denorm is IEEE (both preserve the subnormal). Dynamic mode is
18470 // also rejected since the runtime value is unknown.
18471 bool AllowInaccuracy = N->getFlags().hasApproximateFuncs() &&
18472 FMA->getFlags().hasApproximateFuncs();
18473 if (!AllowInaccuracy) {
18474 const MachineFunction &MF = DAG.getMachineFunction();
18475 DenormalMode Mode = MF.getDenormalMode(APFloat::IEEEsingle());
18476 if (Subtarget->dot2UnconditionalFlush()) {
18477 // gfx90a: fold safe only when f32 denorm flushes.
18479 return SDValue();
18480 } else {
18481 // All other GPUs: fold safe only when f32 denorm is IEEE.
18482 if (Mode != DenormalMode::getIEEE())
18483 return SDValue();
18484 }
18485 }
18486
18487 // fp-contract allows reassociating the fma tree into a dot product.
18488 if (N->getFlags().hasAllowContract() && FMA->getFlags().hasAllowContract()) {
18489 Op1 = Op1.getOperand(0);
18490 Op2 = Op2.getOperand(0);
18491 if (Op1.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
18493 return SDValue();
18494
18495 SDValue Vec1 = Op1.getOperand(0);
18496 SDValue Idx1 = Op1.getOperand(1);
18497 SDValue Vec2 = Op2.getOperand(0);
18498
18499 SDValue FMAOp1 = FMA.getOperand(0);
18500 SDValue FMAOp2 = FMA.getOperand(1);
18501 SDValue FMAAcc = FMA.getOperand(2);
18502
18503 if (FMAOp1.getOpcode() != ISD::FP_EXTEND ||
18504 FMAOp2.getOpcode() != ISD::FP_EXTEND)
18505 return SDValue();
18506
18507 FMAOp1 = FMAOp1.getOperand(0);
18508 FMAOp2 = FMAOp2.getOperand(0);
18509 if (FMAOp1.getOpcode() != ISD::EXTRACT_VECTOR_ELT ||
18511 return SDValue();
18512
18513 SDValue Vec3 = FMAOp1.getOperand(0);
18514 SDValue Vec4 = FMAOp2.getOperand(0);
18515 SDValue Idx2 = FMAOp1.getOperand(1);
18516
18517 if (Idx1 != Op2.getOperand(1) || Idx2 != FMAOp2.getOperand(1))
18518 return SDValue();
18519
18520 if (!isa<ConstantSDNode>(Idx1) || !isa<ConstantSDNode>(Idx2) ||
18521 Idx1 == Idx2)
18522 return SDValue();
18523
18524 if (Vec1 == Vec2 || Vec3 == Vec4)
18525 return SDValue();
18526
18527 if (Vec1.getValueType() != MVT::v2f16 || Vec2.getValueType() != MVT::v2f16)
18528 return SDValue();
18529
18530 if ((Vec1 == Vec3 && Vec2 == Vec4) || (Vec1 == Vec4 && Vec2 == Vec3)) {
18531 return DAG.getNode(AMDGPUISD::FDOT2, SL, MVT::f32, Vec1, Vec2, FMAAcc,
18532 DAG.getTargetConstant(0, SL, MVT::i1));
18533 }
18534 }
18535 return SDValue();
18536}
18537
18538// Given a double-precision ordered or unordered comparison, return the
18539// condition code for an equivalent integral comparison of the operands' upper
18540// 32 bits, or `SETCC_INVALID` if not possible.
18541// For simplicity, no simplification occurs if the operands are not both known
18542// to have sign bit zero.
18543//
18544// EQ/NE:
18545// If LHS.lo32 == RHS.lo32:
18546// setcc LHS, RHS, eq/ne => setcc LHS.hi32, RHS.hi32, eq/ne
18547// If LHS.lo32 != RHS.lo32:
18548// setcc LHS, RHS, eq/ne => setcc LHS.hi32, RHS.hi32, false/true
18549// The reduction is not possible if operands may be +0 and -0.
18550// For ordered eq / unordered ne, at most one operand may be NaN.
18551// For unordered eq / ordered ne, neither operand can be NaN.
18552//
18553// LT/GE:
18554// If LHS.lo32 >= RHS.lo32 (unsigned):
18555// setcc LHS, RHS, [u]lt/ge => LHS.hi32, RHS.hi32, [u]lt/ge
18556// If LHS.lo32 < RHS.lo32 (unsigned):
18557// setcc LHS, RHS, [u]lt/ge => LHS.hi32, RHS.hi32, [u]le/gt
18558// The reduction is only supported if both operands are nonnegative.
18559// For ordered lt / unordered ge, the RHS cannot be NaN.
18560// For unordered lt / ordered ge, neither operand can be NaN.
18561//
18562// LE/GT:
18563// If LHS.lo32 > RHS.lo32 (unsigned):
18564// setcc LHS, RHS, [u]le/gt => LHS.hi32, RHS.hi32, [u]lt/ge
18565// If LHS.lo32 <= RHS.lo32 (unsigned):
18566// setcc LHS, RHS, [u]le/gt => LHS.hi32, RHS.hi32, [u]le/gt
18567// The reduction is only supported if both operands are nonnegative.
18568// For unordered le / ordered gt, the LHS cannot be NaN.
18569// For ordered le / unordered gt, neither operand can be NaN.
18571 const SDValue LHS,
18572 const SDValue RHS,
18573 const SelectionDAG &DAG) {
18574 EVT VT = LHS.getValueType();
18575 assert(VT == MVT::f64 && "Incorrect operand type!");
18576
18577 const KnownBits RHSBits = DAG.computeKnownBits(RHS);
18578 // Bail if RHS sign bit is not known to be zero.
18579 if (!RHSBits.Zero.isSignBitSet())
18580 return ISD::SETCC_INVALID;
18581
18582 const KnownBits RHSKnownLo32 = RHSBits.trunc(32);
18583 const KnownFPClass RHSFPClass =
18585 const bool RHSMaybeNaN = !RHSFPClass.isKnownNeverNaN();
18586
18587 const KnownBits LHSBits = DAG.computeKnownBits(LHS);
18588 const KnownBits LHSKnownLo32 = LHSBits.trunc(32);
18589 const KnownFPClass LHSFPClass =
18591 const bool LHSMaybeNaN = !LHSFPClass.isKnownNeverNaN();
18592
18593 // Bail if LHS sign bit is not known to be zero.
18594 if (!LHSBits.Zero.isSignBitSet())
18595 return ISD::SETCC_INVALID;
18596
18597 switch (CC) {
18598 default:
18599 break;
18600 case ISD::SETEQ:
18601 case ISD::SETOEQ:
18602 case ISD::SETUEQ:
18603 case ISD::SETONE:
18604 case ISD::SETUNE: {
18605 // OEQ should be false if either operand is NaN, so it suffices that at
18606 // least one operand is not NaN.
18607 if (CC == ISD::SETOEQ && LHSMaybeNaN && RHSMaybeNaN)
18608 break;
18609 // UEQ should be true if either operand is NaN, but this cannot be checked
18610 // on underlying bits.
18611 if (CC == ISD::SETUEQ && (LHSMaybeNaN || RHSMaybeNaN))
18612 break;
18613 // ONE should be false if either operand is NaN, but this cannot be
18614 // checked on underlying bits.
18615 if (CC == ISD::SETONE && (LHSMaybeNaN || RHSMaybeNaN))
18616 break;
18617 // UNE should be true if either operand is NaN, so it suffices that they
18618 // are not both NaN.
18619 if (CC == ISD::SETUNE && LHSMaybeNaN && RHSMaybeNaN)
18620 break;
18621
18622 const std::optional<bool> KnownEq =
18623 KnownBits::eq(LHSKnownLo32, RHSKnownLo32);
18624
18625 if (!KnownEq)
18626 break;
18627
18628 if (*KnownEq)
18629 return (CC == ISD::SETEQ || CC == ISD::SETOEQ || CC == ISD::SETUEQ)
18630 ? ISD::SETEQ
18631 : ISD::SETNE;
18632
18633 return (CC == ISD::SETEQ || CC == ISD::SETOEQ || CC == ISD::SETUEQ)
18635 : ISD::SETTRUE;
18636 }
18637 case ISD::SETLT:
18638 case ISD::SETOLT:
18639 case ISD::SETULT:
18640 case ISD::SETGE:
18641 case ISD::SETOGE:
18642 case ISD::SETUGE: {
18643 // OLT should be false if either operand is NaN.
18644 // Since NaNs have maximum exponent and nonzero mantissa, false positives
18645 // are only possible if the RHS is NaN. (No issue with RHS == +inf since
18646 // the inequality is strict)
18647 if (CC == ISD::SETOLT && RHSMaybeNaN)
18648 break;
18649 // ULT should be true if either operand is NaN, but this cannot be ensured
18650 // with a truncated comparison.
18651 if (CC == ISD::SETULT && (LHSMaybeNaN || RHSMaybeNaN))
18652 break;
18653 // OGE should be false if either operand is NaN, but this cannot be
18654 // ensured with a truncated comparison.
18655 if (CC == ISD::SETOGE && (LHSMaybeNaN || RHSMaybeNaN))
18656 break;
18657 // UGE should be true if either operand is NaN.
18658 // False negatives are only possible if the RHS is NaN.
18659 // (No issue with RHS == +inf since the inequality is inclusive)
18660 if (CC == ISD::SETUGE && RHSMaybeNaN)
18661 break;
18662
18663 const std::optional<bool> KnownUge =
18664 KnownBits::uge(LHSKnownLo32, RHSKnownLo32);
18665
18666 if (!KnownUge)
18667 break;
18668
18669 if (*KnownUge) {
18670 // LHS.lo32 uge RHS.lo32, so LHS >= RHS iff LHS.hi32 >= RHS.hi32
18671 return (CC == ISD::SETLT || CC == ISD::SETOLT || CC == ISD::SETULT)
18672 ? ISD::SETLT
18673 : ISD::SETGE;
18674 }
18675 // LHS.lo32 ult RHS.lo32, so LHS >= RHS iff LHS.hi32 > RHS.hi32
18676 return (CC == ISD::SETLT || CC == ISD::SETOLT || CC == ISD::SETULT)
18677 ? ISD::SETLE
18678 : ISD::SETGT;
18679 }
18680 case ISD::SETLE:
18681 case ISD::SETOLE:
18682 case ISD::SETULE:
18683 case ISD::SETGT:
18684 case ISD::SETOGT:
18685 case ISD::SETUGT: {
18686 // OLE should be false if either operand is NaN, but this cannot be
18687 // ensured with a truncated comparison.
18688 if (CC == ISD::SETOLE && (LHSMaybeNaN || RHSMaybeNaN))
18689 break;
18690 // ULE should be true if either operand is NaN.
18691 // False negatives are only possible if the LHS is NaN.
18692 // (No issue with LHS == +inf since the inequality is inclusive)
18693 if (CC == ISD::SETULE && LHSMaybeNaN)
18694 break;
18695 // OGT should be false if either operand is NaN.
18696 // False positives are only possible if the LHS is NaN.
18697 // (No issue with LHS == +inf since the inequality is strict)
18698 if (CC == ISD::SETOGT && LHSMaybeNaN)
18699 break;
18700 // UGT should be true if either operand is NaN, but this cannot be ensured
18701 // with a truncated comparison.
18702 if (CC == ISD::SETUGT && (LHSMaybeNaN || RHSMaybeNaN))
18703 break;
18704
18705 const std::optional<bool> KnownUle =
18706 KnownBits::ule(LHSKnownLo32, RHSKnownLo32);
18707
18708 if (!KnownUle)
18709 break;
18710
18711 if (*KnownUle) {
18712 // LHS.lo32 ule RHS.lo32, so LHS <= RHS iff LHS.hi32 <= RHS.hi32
18713 return (CC == ISD::SETLE || CC == ISD::SETOLE || CC == ISD::SETULE)
18714 ? ISD::SETLE
18715 : ISD::SETGT;
18716 }
18717 // LHS.lo32 ugt RHS.lo32, so LHS <= RHS iff LHS.hi32 < RHS.hi32
18718 return (CC == ISD::SETLE || CC == ISD::SETOLE || CC == ISD::SETULE)
18719 ? ISD::SETLT
18720 : ISD::SETGE;
18721 }
18722 }
18723
18724 return ISD::SETCC_INVALID;
18725}
18726
18727SDValue SITargetLowering::performSetCCCombine(SDNode *N,
18728 DAGCombinerInfo &DCI) const {
18729 SelectionDAG &DAG = DCI.DAG;
18730 SDLoc SL(N);
18731
18732 SDValue LHS = N->getOperand(0);
18733 SDValue RHS = N->getOperand(1);
18734 EVT VT = LHS.getValueType();
18735 ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
18736
18737 auto *CRHS = dyn_cast<ConstantSDNode>(RHS);
18738 if (!CRHS) {
18740 if (CRHS) {
18741 std::swap(LHS, RHS);
18742 CC = getSetCCSwappedOperands(CC);
18743 }
18744 }
18745
18746 if (CRHS) {
18747 if (VT == MVT::i32 && LHS.getOpcode() == ISD::SIGN_EXTEND &&
18748 isBoolSGPR(LHS.getOperand(0))) {
18749 // setcc (sext from i1 cc), -1, ne|sgt|ult) => not cc => xor cc, -1
18750 // setcc (sext from i1 cc), -1, eq|sle|uge) => cc
18751 // setcc (sext from i1 cc), 0, eq|sge|ule) => not cc => xor cc, -1
18752 // setcc (sext from i1 cc), 0, ne|ugt|slt) => cc
18753 if ((CRHS->isAllOnes() &&
18754 (CC == ISD::SETNE || CC == ISD::SETGT || CC == ISD::SETULT)) ||
18755 (CRHS->isZero() &&
18756 (CC == ISD::SETEQ || CC == ISD::SETGE || CC == ISD::SETULE)))
18757 return DAG.getNode(ISD::XOR, SL, MVT::i1, LHS.getOperand(0),
18758 DAG.getAllOnesConstant(SL, MVT::i1));
18759 if ((CRHS->isAllOnes() &&
18760 (CC == ISD::SETEQ || CC == ISD::SETLE || CC == ISD::SETUGE)) ||
18761 (CRHS->isZero() &&
18762 (CC == ISD::SETNE || CC == ISD::SETUGT || CC == ISD::SETLT)))
18763 return LHS.getOperand(0);
18764 }
18765
18766 const APInt &CRHSVal = CRHS->getAPIntValue();
18767 if ((CC == ISD::SETEQ || CC == ISD::SETNE) &&
18768 LHS.getOpcode() == ISD::SELECT &&
18769 isa<ConstantSDNode>(LHS.getOperand(1)) &&
18770 isa<ConstantSDNode>(LHS.getOperand(2)) &&
18771 isBoolSGPR(LHS.getOperand(0))) {
18772 // Given CT != FT:
18773 // setcc (select cc, CT, CF), CF, eq => xor cc, -1
18774 // setcc (select cc, CT, CF), CF, ne => cc
18775 // setcc (select cc, CT, CF), CT, ne => xor cc, -1
18776 // setcc (select cc, CT, CF), CT, eq => cc
18777 const APInt &CT = LHS.getConstantOperandAPInt(1);
18778 const APInt &CF = LHS.getConstantOperandAPInt(2);
18779
18780 if (CT != CF) {
18781 if ((CF == CRHSVal && CC == ISD::SETEQ) ||
18782 (CT == CRHSVal && CC == ISD::SETNE))
18783 return DAG.getNOT(SL, LHS.getOperand(0), MVT::i1);
18784 if ((CF == CRHSVal && CC == ISD::SETNE) ||
18785 (CT == CRHSVal && CC == ISD::SETEQ))
18786 return LHS.getOperand(0);
18787 }
18788 }
18789 }
18790
18791 // Truncate 64-bit setcc to test only upper 32-bits of its operands in the
18792 // following cases where information about the lower 32-bits of its operands
18793 // is known:
18794 //
18795 // If LHS.lo32 == RHS.lo32:
18796 // setcc LHS, RHS, eq/ne => setcc LHS.hi32, RHS.hi32, eq/ne
18797 // If LHS.lo32 != RHS.lo32:
18798 // setcc LHS, RHS, eq/ne => setcc LHS.hi32, RHS.hi32, false/true
18799 // If LHS.lo32 >= RHS.lo32 (unsigned):
18800 // setcc LHS, RHS, [u]lt/ge => LHS.hi32, RHS.hi32, [u]lt/ge
18801 // If LHS.lo32 > RHS.lo32 (unsigned):
18802 // setcc LHS, RHS, [u]le/gt => LHS.hi32, RHS.hi32, [u]lt/ge
18803 // If LHS.lo32 <= RHS.lo32 (unsigned):
18804 // setcc LHS, RHS, [u]le/gt => LHS.hi32, RHS.hi32, [u]le/gt
18805 // If LHS.lo32 < RHS.lo32 (unsigned):
18806 // setcc LHS, RHS, [u]lt/ge => LHS.hi32, RHS.hi32, [u]le/gt
18807 if (VT == MVT::i64) {
18808 const KnownBits LHSKnownLo32 = DAG.computeKnownBits(LHS).trunc(32);
18809 const KnownBits RHSKnownLo32 = DAG.computeKnownBits(RHS).trunc(32);
18810
18811 // NewCC is valid iff we can truncate the setcc to only test the upper 32
18812 // bits
18814
18815 switch (CC) {
18816 default:
18817 break;
18818 case ISD::SETEQ: {
18819 const std::optional<bool> KnownEq =
18820 KnownBits::eq(LHSKnownLo32, RHSKnownLo32);
18821 if (KnownEq)
18822 NewCC = *KnownEq ? ISD::SETEQ : ISD::SETFALSE;
18823
18824 break;
18825 }
18826 case ISD::SETNE: {
18827 const std::optional<bool> KnownEq =
18828 KnownBits::eq(LHSKnownLo32, RHSKnownLo32);
18829 if (KnownEq)
18830 NewCC = *KnownEq ? ISD::SETNE : ISD::SETTRUE;
18831
18832 break;
18833 }
18834 case ISD::SETULT:
18835 case ISD::SETUGE:
18836 case ISD::SETLT:
18837 case ISD::SETGE: {
18838 const std::optional<bool> KnownUge =
18839 KnownBits::uge(LHSKnownLo32, RHSKnownLo32);
18840 if (KnownUge) {
18841 if (*KnownUge) {
18842 // LHS.lo32 uge RHS.lo32, so LHS >= RHS iff LHS.hi32 >= RHS.hi32
18843 NewCC = CC;
18844 } else {
18845 // LHS.lo32 ult RHS.lo32, so LHS >= RHS iff LHS.hi32 > RHS.hi32
18846 NewCC = CC == ISD::SETULT ? ISD::SETULE
18847 : CC == ISD::SETUGE ? ISD::SETUGT
18848 : CC == ISD::SETLT ? ISD::SETLE
18849 : ISD::SETGT;
18850 }
18851 }
18852 break;
18853 }
18854 case ISD::SETULE:
18855 case ISD::SETUGT:
18856 case ISD::SETLE:
18857 case ISD::SETGT: {
18858 const std::optional<bool> KnownUle =
18859 KnownBits::ule(LHSKnownLo32, RHSKnownLo32);
18860 if (KnownUle) {
18861 if (*KnownUle) {
18862 // LHS.lo32 ule RHS.lo32, so LHS <= RHS iff LHS.hi32 <= RHS.hi32
18863 NewCC = CC;
18864 } else {
18865 // LHS.lo32 ugt RHS.lo32, so LHS <= RHS iff LHS.hi32 < RHS.hi32
18866 NewCC = CC == ISD::SETULE ? ISD::SETULT
18867 : CC == ISD::SETUGT ? ISD::SETUGE
18868 : CC == ISD::SETLE ? ISD::SETLT
18869 : ISD::SETGE;
18870 }
18871 }
18872 break;
18873 }
18874 }
18875
18876 if (NewCC != ISD::SETCC_INVALID)
18877 return DAG.getSetCC(SL, N->getValueType(0), getHiHalf64(LHS, DAG),
18878 getHiHalf64(RHS, DAG), NewCC);
18879 }
18880
18881 // Eliminate setcc by using carryout from add/sub instruction
18882
18883 // LHS = ADD i64 RHS, Z LHSlo = UADDO i32 RHSlo, Zlo
18884 // setcc LHS ult RHS -> LHSHi = UADDO_CARRY i32 RHShi, Zhi
18885 // similarly for subtraction
18886
18887 // LHS = ADD i64 Y, 1 LHSlo = UADDO i32 Ylo, 1
18888 // setcc LHS eq 0 -> LHSHi = UADDO_CARRY i32 Yhi, 0
18889
18890 if (VT == MVT::i64 && ((CC == ISD::SETULT &&
18892 (CC == ISD::SETUGT &&
18894 (CC == ISD::SETEQ && CRHS && CRHS->isZero() &&
18895 sd_match(LHS, m_Add(m_Value(), m_One()))))) {
18896 bool IsAdd = LHS.getOpcode() == ISD::ADD;
18897
18898 SDValue Op0 = LHS.getOperand(0);
18899 SDValue Op1 = LHS.getOperand(1);
18900
18901 SDValue Op0Lo = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Op0);
18902 SDValue Op1Lo = DAG.getNode(ISD::TRUNCATE, SL, MVT::i32, Op1);
18903
18904 SDValue Op0Hi = getHiHalf64(Op0, DAG);
18905 SDValue Op1Hi = getHiHalf64(Op1, DAG);
18906
18907 SDValue NodeLo =
18908 DAG.getNode(IsAdd ? ISD::UADDO : ISD::USUBO, SL,
18909 DAG.getVTList(MVT::i32, MVT::i1), {Op0Lo, Op1Lo});
18910
18911 SDValue CarryInHi = NodeLo.getValue(1);
18912 SDValue NodeHi = DAG.getNode(IsAdd ? ISD::UADDO_CARRY : ISD::USUBO_CARRY,
18913 SL, DAG.getVTList(MVT::i32, MVT::i1),
18914 {Op0Hi, Op1Hi, CarryInHi});
18915
18916 SDValue ResultLo = NodeLo.getValue(0);
18917 SDValue ResultHi = NodeHi.getValue(0);
18918
18919 SDValue JoinedResult =
18920 DAG.getBuildVector(MVT::v2i32, SL, {ResultLo, ResultHi});
18921
18922 SDValue Result = DAG.getNode(ISD::BITCAST, SL, VT, JoinedResult);
18923 SDValue Overflow = NodeHi.getValue(1);
18924 DCI.CombineTo(LHS.getNode(), Result);
18925 return Overflow;
18926 }
18927
18928 if (VT != MVT::f32 && VT != MVT::f64 &&
18929 (!Subtarget->has16BitInsts() || VT != MVT::f16))
18930 return SDValue();
18931
18932 // Match isinf/isfinite pattern
18933 // (fcmp oeq (fabs x), inf) -> (fp_class x, (p_infinity | n_infinity))
18934 // (fcmp one (fabs x), inf) -> (fp_class x,
18935 // (p_normal | n_normal | p_subnormal | n_subnormal | p_zero | n_zero)
18936 if ((CC == ISD::SETOEQ || CC == ISD::SETONE) &&
18937 LHS.getOpcode() == ISD::FABS) {
18938 const ConstantFPSDNode *CRHS = dyn_cast<ConstantFPSDNode>(RHS);
18939 if (!CRHS)
18940 return SDValue();
18941
18942 const APFloat &APF = CRHS->getValueAPF();
18943 if (APF.isInfinity() && !APF.isNegative()) {
18944 const unsigned IsInfMask =
18946 const unsigned IsFiniteMask =
18950 unsigned Mask = CC == ISD::SETOEQ ? IsInfMask : IsFiniteMask;
18951 return DAG.getNode(AMDGPUISD::FP_CLASS, SL, MVT::i1, LHS.getOperand(0),
18952 DAG.getConstant(Mask, SL, MVT::i32));
18953 }
18954 }
18955
18956 if (VT == MVT::f64) {
18957 ISD::CondCode HiHalfCC = tryReduceF64CompareToHiHalf(CC, LHS, RHS, DAG);
18958 if (HiHalfCC != ISD::SETCC_INVALID)
18959 return DAG.getSetCC(SL, N->getValueType(0), getHiHalf64(LHS, DAG),
18960 getHiHalf64(RHS, DAG), HiHalfCC);
18961 }
18962
18963 return SDValue();
18964}
18965
18966SDValue
18967SITargetLowering::performCvtF32UByteNCombine(SDNode *N,
18968 DAGCombinerInfo &DCI) const {
18969 SelectionDAG &DAG = DCI.DAG;
18970 SDLoc SL(N);
18971 unsigned Offset = N->getOpcode() - AMDGPUISD::CVT_F32_UBYTE0;
18972
18973 SDValue Src = N->getOperand(0);
18974 SDValue Shift = N->getOperand(0);
18975
18976 // TODO: Extend type shouldn't matter (assuming legal types).
18977 if (Shift.getOpcode() == ISD::ZERO_EXTEND)
18978 Shift = Shift.getOperand(0);
18979
18980 if (Shift.getOpcode() == ISD::SRL || Shift.getOpcode() == ISD::SHL) {
18981 // cvt_f32_ubyte1 (shl x, 8) -> cvt_f32_ubyte0 x
18982 // cvt_f32_ubyte3 (shl x, 16) -> cvt_f32_ubyte1 x
18983 // cvt_f32_ubyte0 (srl x, 16) -> cvt_f32_ubyte2 x
18984 // cvt_f32_ubyte1 (srl x, 16) -> cvt_f32_ubyte3 x
18985 // cvt_f32_ubyte0 (srl x, 8) -> cvt_f32_ubyte1 x
18986 if (auto *C = dyn_cast<ConstantSDNode>(Shift.getOperand(1))) {
18987 SDValue Shifted = DAG.getZExtOrTrunc(
18988 Shift.getOperand(0), SDLoc(Shift.getOperand(0)), MVT::i32);
18989
18990 unsigned ShiftOffset = 8 * Offset;
18991 if (Shift.getOpcode() == ISD::SHL)
18992 ShiftOffset -= C->getZExtValue();
18993 else
18994 ShiftOffset += C->getZExtValue();
18995
18996 if (ShiftOffset < 32 && (ShiftOffset % 8) == 0) {
18997 return DAG.getNode(AMDGPUISD::CVT_F32_UBYTE0 + ShiftOffset / 8, SL,
18998 MVT::f32, Shifted);
18999 }
19000 }
19001 }
19002
19003 const TargetLowering &TLI = DAG.getTargetLoweringInfo();
19004 APInt DemandedBits = APInt::getBitsSet(32, 8 * Offset, 8 * Offset + 8);
19005 if (TLI.SimplifyDemandedBits(Src, DemandedBits, DCI)) {
19006 // We simplified Src. If this node is not dead, visit it again so it is
19007 // folded properly.
19008 if (N->getOpcode() != ISD::DELETED_NODE)
19009 DCI.AddToWorklist(N);
19010 return SDValue(N, 0);
19011 }
19012
19013 // Handle (or x, (srl y, 8)) pattern when known bits are zero.
19014 if (SDValue DemandedSrc =
19015 TLI.SimplifyMultipleUseDemandedBits(Src, DemandedBits, DAG))
19016 return DAG.getNode(N->getOpcode(), SL, MVT::f32, DemandedSrc);
19017
19018 return SDValue();
19019}
19020
19021SDValue SITargetLowering::performClampCombine(SDNode *N,
19022 DAGCombinerInfo &DCI) const {
19023 ConstantFPSDNode *CSrc = dyn_cast<ConstantFPSDNode>(N->getOperand(0));
19024 if (!CSrc)
19025 return SDValue();
19026
19027 const MachineFunction &MF = DCI.DAG.getMachineFunction();
19028 const APFloat &F = CSrc->getValueAPF();
19029 APFloat Zero = APFloat::getZero(F.getSemantics());
19030 if (F < Zero ||
19031 (F.isNaN() && MF.getInfo<SIMachineFunctionInfo>()->getMode().DX10Clamp)) {
19032 return DCI.DAG.getConstantFP(Zero, SDLoc(N), N->getValueType(0));
19033 }
19034
19035 APFloat One = APFloat::getOne(F.getSemantics());
19036 if (F > One)
19037 return DCI.DAG.getConstantFP(One, SDLoc(N), N->getValueType(0));
19038
19039 return getCanonicalConstantFP(DCI.DAG, SDLoc(N), N->getValueType(0), F);
19040}
19041
19042// Check if V is the exponent result of a frexp operation. Returns the frexp
19043// input via FrexpInput if matched. We only match the exponent (not mantissa)
19044// because V_FREXP_MANT returns its input for Inf/NaN, not zero.
19045static bool isFrexpExp(SDValue V, SDValue &FrexpInput) {
19046 // ISD::FFREXP returns {mant, exp} - only match if using the exp result
19047 // (result number 1).
19048 if (V.getOpcode() == ISD::FFREXP && V.getResNo() == 1) {
19049 FrexpInput = V.getOperand(0);
19050 return true;
19051 }
19053 m_Value(FrexpInput))))
19054 return true;
19055 return false;
19056}
19057
19058SDValue
19059SITargetLowering::performFrexpSelectCombine(SDNode *N,
19060 DAGCombinerInfo &DCI) const {
19061 // This optimization only applies when the hardware handles inf/nan correctly.
19062 if (Subtarget->hasFractBug())
19063 return SDValue();
19064
19065 SDValue Cond = N->getOperand(0);
19066 SDValue TrueVal = N->getOperand(1);
19067 SDValue FalseVal = N->getOperand(2);
19068
19069 // Identify which operand is the frexp result and which is the zero constant.
19070 // Pattern 1: select cond, 0, frexp_result (cond true -> return 0)
19071 // Pattern 2: select cond, frexp_result, 0 (cond false -> return 0)
19072 SDValue FrexpVal;
19073 SDValue ZeroVal;
19074 bool CondSelectsZero; // If true, condition=true selects zero
19075
19076 // Check if FrexpVal comes from ISD::FFREXP (exponent result only) or
19077 // amdgcn_frexp_exp intrinsic.
19078 SDValue FrexpInput;
19079 if (isFrexpExp(FalseVal, FrexpInput)) {
19080 FrexpVal = FalseVal;
19081 ZeroVal = TrueVal;
19082 CondSelectsZero = true;
19083 } else if (isFrexpExp(TrueVal, FrexpInput)) {
19084 FrexpVal = TrueVal;
19085 ZeroVal = FalseVal;
19086 CondSelectsZero = false;
19087 } else {
19088 return SDValue();
19089 }
19090
19091 // frexp_exp returns integer, so check for integer zero.
19092 if (!isNullConstant(ZeroVal))
19093 return SDValue();
19094
19095 // The frexp intrinsics ignore sign, so we can strip sign ops when comparing.
19096 SDValue FrexpInputStripped = peekFPSignOps(FrexpInput);
19097
19098 bool IsNonFiniteTest = false;
19099
19100 // Handle SETCC conditions for inf/nan tests.
19101 // The canonical form of these checks is fcmp + fabs.
19102 if (Cond.getOpcode() == ISD::SETCC) {
19103 ISD::CondCode CC = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
19104 SDValue CondLHS = Cond.getOperand(0);
19105 SDValue CondRHS = Cond.getOperand(1);
19106
19107 // Check if LHS is fabs(FrexpInput) - required for infinity comparisons.
19108 SDValue FAbsInput;
19109 bool LHSIsFabs = sd_match(CondLHS, m_FAbs(m_Value(FAbsInput)));
19110 bool LHSMatchesFrexp =
19111 (CondLHS == FrexpInput) ||
19112 (LHSIsFabs && peekFPSignOps(FAbsInput) == FrexpInputStripped) ||
19113 (peekFPSignOps(CondLHS) == FrexpInputStripped);
19114 bool RHSMatchesFrexp = (CondRHS == FrexpInput) ||
19115 (peekFPSignOps(CondRHS) == FrexpInputStripped);
19116
19117 if (CC == ISD::SETUO) {
19118 // fcmp uno x, y - true if either x or y is NaN
19119 // We can only fold if the non-frexp operand is known to never be NaN,
19120 // otherwise the comparison could be true due to the other operand.
19121 // Special case: fcmp uno x, x (same operand) is a valid NaN test.
19122 SelectionDAG &DAG = DCI.DAG;
19123 if (LHSMatchesFrexp &&
19124 (CondLHS == CondRHS || DAG.isKnownNeverNaN(CondRHS)))
19125 IsNonFiniteTest = CondSelectsZero;
19126 else if (RHSMatchesFrexp && DAG.isKnownNeverNaN(CondLHS))
19127 IsNonFiniteTest = CondSelectsZero;
19128 } else if ((CC == ISD::SETOEQ || CC == ISD::SETUEQ) && LHSMatchesFrexp &&
19129 LHSIsFabs &&
19130 sd_match(CondRHS,
19132 CondRHS.getValueType().getFltSemantics())))) {
19133 // fcmp oeq/ueq fabs(x), +inf - true if x is inf (or inf/nan for ueq)
19134 IsNonFiniteTest = CondSelectsZero;
19135 } else if ((CC == ISD::SETONE || CC == ISD::SETUNE) && LHSMatchesFrexp &&
19136 LHSIsFabs &&
19137 sd_match(CondRHS,
19139 CondRHS.getValueType().getFltSemantics())))) {
19140 // fcmp one/une fabs(x), +inf - true if x is NOT inf
19141 IsNonFiniteTest = !CondSelectsZero;
19142 } else if (CC == ISD::SETO) {
19143 // fcmp ord x, y - true if both are NOT NaN
19144 // We can only fold if the non-frexp operand is known to never be NaN,
19145 // otherwise the comparison could be false due to the other operand.
19146 // Special case: fcmp ord x, x (same operand) is a valid not-NaN test.
19147 SelectionDAG &DAG = DCI.DAG;
19148 if (LHSMatchesFrexp &&
19149 (CondLHS == CondRHS || DAG.isKnownNeverNaN(CondRHS)))
19150 IsNonFiniteTest = !CondSelectsZero;
19151 else if (RHSMatchesFrexp && DAG.isKnownNeverNaN(CondLHS))
19152 IsNonFiniteTest = !CondSelectsZero;
19153 }
19154 }
19155
19156 if (!IsNonFiniteTest)
19157 return SDValue();
19158
19159 // The select can be eliminated - just return the frexp result directly.
19160 return FrexpVal;
19161}
19162
19163SDValue SITargetLowering::performSelectCombine(SDNode *N,
19164 DAGCombinerInfo &DCI) const {
19165
19166 // Try to fold CMP + SELECT patterns with shared constants (both FP and
19167 // integer).
19168 // Detect when CMP and SELECT use the same constant and fold them to avoid
19169 // loading the constant twice. Specifically handles patterns like:
19170 // %cmp = icmp eq i32 %val, 4242
19171 // %sel = select i1 %cmp, i32 4242, i32 %other
19172 // It can be optimized to reuse %val instead of 4242 in select.
19173 SDValue Cond = N->getOperand(0);
19174 SDValue TrueVal = N->getOperand(1);
19175 SDValue FalseVal = N->getOperand(2);
19176
19177 // Check if condition is a comparison.
19178 if (Cond.getOpcode() != ISD::SETCC)
19179 return SDValue();
19180
19181 SDValue LHS = Cond.getOperand(0);
19182 SDValue RHS = Cond.getOperand(1);
19183 ISD::CondCode CC = cast<CondCodeSDNode>(Cond.getOperand(2))->get();
19184
19185 bool isFloatingPoint = LHS.getValueType().isFloatingPoint();
19186 bool isInteger = LHS.getValueType().isInteger();
19187
19188 // Handle simple floating-point and integer types only.
19189 if (!isFloatingPoint && !isInteger)
19190 return SDValue();
19191
19192 // Bare SETEQ/SETNE is the builder's NaN-impossible downgrade.
19193 bool isEquality = CC == ISD::SETEQ || (isFloatingPoint && CC == ISD::SETOEQ);
19194 bool isNonEquality =
19195 CC == ISD::SETNE || (isFloatingPoint && CC == ISD::SETONE);
19196 if (!isEquality && !isNonEquality)
19197 return SDValue();
19198
19199 SDValue ArgVal, ConstVal;
19200 if ((isFloatingPoint && isa<ConstantFPSDNode>(RHS)) ||
19201 (isInteger && isa<ConstantSDNode>(RHS))) {
19202 ConstVal = RHS;
19203 ArgVal = LHS;
19204 } else if ((isFloatingPoint && isa<ConstantFPSDNode>(LHS)) ||
19205 (isInteger && isa<ConstantSDNode>(LHS))) {
19206 ConstVal = LHS;
19207 ArgVal = RHS;
19208 } else {
19209 return SDValue();
19210 }
19211
19212 // Skip optimization for inlinable immediates.
19213 if (isFloatingPoint) {
19214 const APFloat &Val = cast<ConstantFPSDNode>(ConstVal)->getValueAPF();
19215 if (!Val.isNormal() || Subtarget->getInstrInfo()->isInlineConstant(Val))
19216 return SDValue();
19217 } else {
19218 const std::optional<int64_t> Val =
19219 cast<ConstantSDNode>(ConstVal)->getAPIntValue().trySExtValue();
19220 if (Val && AMDGPU::isInlinableIntLiteral(*Val))
19221 return SDValue();
19222 }
19223
19224 // For equality and non-equality comparisons, patterns:
19225 // select (setcc x, const), const, y -> select (setcc x, const), x, y
19226 // select (setccinv x, const), y, const -> select (setccinv x, const), y, x
19227 if (!(isEquality && TrueVal == ConstVal) &&
19228 !(isNonEquality && FalseVal == ConstVal))
19229 return SDValue();
19230
19231 // SETONE's false arm is also taken for NaN ArgVal, so require NaN excluded.
19232 if (isFloatingPoint && isNonEquality && FalseVal == ConstVal &&
19233 !Cond->getFlags().hasNoNaNs() && !DCI.DAG.isKnownNeverNaN(ArgVal))
19234 return SDValue();
19235
19236 SDValue SelectLHS = (isEquality && TrueVal == ConstVal) ? ArgVal : TrueVal;
19237 SDValue SelectRHS =
19238 (isNonEquality && FalseVal == ConstVal) ? ArgVal : FalseVal;
19239 return DCI.DAG.getNode(ISD::SELECT, SDLoc(N), N->getValueType(0), Cond,
19240 SelectLHS, SelectRHS);
19241}
19242
19244 DAGCombinerInfo &DCI) const {
19245 switch (N->getOpcode()) {
19246 case ISD::ABS:
19247 if (SDValue Res = promoteUniformUnaryOpToI32(SDValue(N, 0), DCI))
19248 return Res;
19249 break;
19250 case ISD::ADD:
19251 case ISD::SUB:
19252 case ISD::SHL:
19253 case ISD::SRL:
19254 case ISD::SRA:
19255 case ISD::AND:
19256 case ISD::OR:
19257 case ISD::XOR:
19258 case ISD::MUL:
19259 case ISD::SETCC:
19260 case ISD::SELECT:
19261 case ISD::SMIN:
19262 case ISD::SMAX:
19263 case ISD::UMIN:
19264 case ISD::UMAX:
19265 case ISD::USUBSAT:
19266 case ISD::UADDSAT:
19267 if (auto Res = promoteUniformOpToI32(SDValue(N, 0), DCI))
19268 return Res;
19269 break;
19270 default:
19271 break;
19272 }
19273
19274 if (getTargetMachine().getOptLevel() == CodeGenOptLevel::None)
19275 return SDValue();
19276
19277 switch (N->getOpcode()) {
19278 case ISD::ADD:
19279 return performAddCombine(N, DCI);
19280 case ISD::PTRADD:
19281 return performPtrAddCombine(N, DCI);
19282 case ISD::SUB:
19283 return performSubCombine(N, DCI);
19284 case ISD::FADD:
19285 return performFAddCombine(N, DCI);
19286 case ISD::FSUB:
19287 return performFSubCombine(N, DCI);
19288 case ISD::FDIV:
19289 return performFDivCombine(N, DCI);
19290 case ISD::FMUL:
19291 return performFMulCombine(N, DCI);
19292 case ISD::SETCC:
19293 return performSetCCCombine(N, DCI);
19294 case ISD::SELECT:
19295 if (auto Res = performFrexpSelectCombine(N, DCI))
19296 return Res;
19297 if (auto Res = performSelectCombine(N, DCI))
19298 return Res;
19299 break;
19300 case ISD::FMAXNUM:
19301 case ISD::FMINNUM:
19302 case ISD::FMAXNUM_IEEE:
19303 case ISD::FMINNUM_IEEE:
19304 case ISD::FMAXIMUM:
19305 case ISD::FMINIMUM:
19306 case ISD::FMAXIMUMNUM:
19307 case ISD::FMINIMUMNUM:
19308 case ISD::SMAX:
19309 case ISD::SMIN:
19310 case ISD::UMAX:
19311 case ISD::UMIN:
19312 case AMDGPUISD::FMIN_LEGACY:
19313 case AMDGPUISD::FMAX_LEGACY:
19314 return performMinMaxCombine(N, DCI);
19315 case ISD::FMA:
19316 return performFMACombine(N, DCI);
19317 case ISD::AND:
19318 return performAndCombine(N, DCI);
19319 case ISD::OR:
19320 return performOrCombine(N, DCI);
19321 case ISD::FSHR: {
19323 if (N->getValueType(0) == MVT::i32 && N->isDivergent() &&
19324 TII->pseudoToMCOpcode(AMDGPU::V_PERM_B32_e64) != -1) {
19325 return matchPERM(N, DCI);
19326 }
19327 break;
19328 }
19329 case ISD::XOR:
19330 return performXorCombine(N, DCI);
19331 case ISD::ANY_EXTEND:
19332 case ISD::ZERO_EXTEND:
19333 return performZeroOrAnyExtendCombine(N, DCI);
19335 return performSignExtendInRegCombine(N, DCI);
19336 case AMDGPUISD::FP_CLASS:
19337 return performClassCombine(N, DCI);
19338 case ISD::FCANONICALIZE:
19339 return performFCanonicalizeCombine(N, DCI);
19340 case AMDGPUISD::RCP:
19341 return performRcpCombine(N, DCI);
19342 case ISD::FLDEXP:
19343 case AMDGPUISD::FRACT:
19344 case AMDGPUISD::RSQ:
19345 case AMDGPUISD::RCP_LEGACY:
19346 case AMDGPUISD::RCP_IFLAG:
19347 case AMDGPUISD::RSQ_CLAMP: {
19348 // FIXME: This is probably wrong. If src is an sNaN, it won't be quieted
19349 SDValue Src = N->getOperand(0);
19350 if (Src.isUndef())
19351 return Src;
19352 break;
19353 }
19354 case ISD::SINT_TO_FP:
19355 case ISD::UINT_TO_FP:
19356 return performUCharToFloatCombine(N, DCI);
19357 case ISD::FCOPYSIGN:
19358 return performFCopySignCombine(N, DCI);
19359 case AMDGPUISD::CVT_F32_UBYTE0:
19360 case AMDGPUISD::CVT_F32_UBYTE1:
19361 case AMDGPUISD::CVT_F32_UBYTE2:
19362 case AMDGPUISD::CVT_F32_UBYTE3:
19363 return performCvtF32UByteNCombine(N, DCI);
19364 case AMDGPUISD::FMED3:
19365 return performFMed3Combine(N, DCI);
19366 case AMDGPUISD::CVT_PKRTZ_F16_F32:
19367 return performCvtPkRTZCombine(N, DCI);
19368 case AMDGPUISD::CLAMP:
19369 return performClampCombine(N, DCI);
19370 case ISD::SCALAR_TO_VECTOR: {
19371 SelectionDAG &DAG = DCI.DAG;
19372 EVT VT = N->getValueType(0);
19373
19374 // When bf16 inline constants live in the upper half of the expanded fp32
19375 // constant, only a splat is encodable as an inline constant. The high lane
19376 // is dead here, so splat it.
19377 if (VT == MVT::v2bf16 && Subtarget->hasBF16InlineConstFromUpperFP32()) {
19378 auto *C = dyn_cast<ConstantFPSDNode>(N->getOperand(0));
19380 C->getValueAPF().bitcastToAPInt().getSExtValue(),
19381 Subtarget->hasInv2PiInlineImm()))
19382 return DAG.getBuildVector(VT, SDLoc(N),
19383 {N->getOperand(0), N->getOperand(0)});
19384 }
19385
19386 // v2i16 (scalar_to_vector i16:x) -> v2i16 (bitcast (any_extend i16:x))
19387 if (VT == MVT::v2i16 || VT == MVT::v2f16 || VT == MVT::v2bf16) {
19388 SDLoc SL(N);
19389 SDValue Src = N->getOperand(0);
19390 EVT EltVT = Src.getValueType();
19391 if (EltVT != MVT::i16)
19392 Src = DAG.getNode(ISD::BITCAST, SL, MVT::i16, Src);
19393
19394 SDValue Ext = DAG.getNode(ISD::ANY_EXTEND, SL, MVT::i32, Src);
19395 return DAG.getNode(ISD::BITCAST, SL, VT, Ext);
19396 }
19397
19398 break;
19399 }
19401 return performExtractVectorEltCombine(N, DCI);
19403 return performInsertVectorEltCombine(N, DCI);
19404 case ISD::FP_ROUND:
19405 return performFPRoundCombine(N, DCI);
19406 case ISD::LOAD: {
19407 if (SDValue Widened = widenLoad(cast<LoadSDNode>(N), DCI))
19408 return Widened;
19409 [[fallthrough]];
19410 }
19411 default: {
19412 if (!DCI.isBeforeLegalize()) {
19413 if (MemSDNode *MemNode = dyn_cast<MemSDNode>(N))
19414 return performMemSDNodeCombine(MemNode, DCI);
19415 }
19416
19417 break;
19418 }
19419 }
19420
19422}
19423
19424/// Helper function for adjustWritemask
19425static unsigned SubIdx2Lane(unsigned Idx) {
19426 switch (Idx) {
19427 default:
19428 return ~0u;
19429 case AMDGPU::sub0:
19430 return 0;
19431 case AMDGPU::sub1:
19432 return 1;
19433 case AMDGPU::sub2:
19434 return 2;
19435 case AMDGPU::sub3:
19436 return 3;
19437 case AMDGPU::sub4:
19438 return 4; // Possible with TFE/LWE
19439 }
19440}
19441
19442/// Adjust the writemask of MIMG, VIMAGE or VSAMPLE instructions
19443SDNode *SITargetLowering::adjustWritemask(MachineSDNode *&Node,
19444 SelectionDAG &DAG) const {
19445 unsigned Opcode = Node->getMachineOpcode();
19446
19447 // Subtract 1 because the vdata output is not a MachineSDNode operand.
19448 int D16Idx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::d16) - 1;
19449 if (D16Idx >= 0 && Node->getConstantOperandVal(D16Idx))
19450 return Node; // not implemented for D16
19451
19452 SDNode *Users[5] = {nullptr};
19453 unsigned Lane = 0;
19454 unsigned DmaskIdx =
19455 AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::dmask) - 1;
19456 unsigned OldDmask = Node->getConstantOperandVal(DmaskIdx);
19457 unsigned NewDmask = 0;
19458 unsigned TFEIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::tfe) - 1;
19459 unsigned LWEIdx = AMDGPU::getNamedOperandIdx(Opcode, AMDGPU::OpName::lwe) - 1;
19460 bool UsesTFC = (int(TFEIdx) >= 0 && Node->getConstantOperandVal(TFEIdx)) ||
19461 (int(LWEIdx) >= 0 && Node->getConstantOperandVal(LWEIdx));
19462 unsigned TFCLane = 0;
19463 bool HasChain = Node->getNumValues() > 1;
19464
19465 if (OldDmask == 0) {
19466 // These are folded out, but on the chance it happens don't assert.
19467 return Node;
19468 }
19469
19470 unsigned OldBitsSet = llvm::popcount(OldDmask);
19471 // Work out which is the TFE/LWE lane if that is enabled.
19472 if (UsesTFC) {
19473 TFCLane = OldBitsSet;
19474 }
19475
19476 // Try to figure out the used register components
19477 for (SDUse &Use : Node->uses()) {
19478
19479 // Don't look at users of the chain.
19480 if (Use.getResNo() != 0)
19481 continue;
19482
19483 SDNode *User = Use.getUser();
19484
19485 // Abort if we can't understand the usage
19486 if (!User->isMachineOpcode() ||
19487 User->getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG)
19488 return Node;
19489
19490 // Lane means which subreg of %vgpra_vgprb_vgprc_vgprd is used.
19491 // Note that subregs are packed, i.e. Lane==0 is the first bit set
19492 // in OldDmask, so it can be any of X,Y,Z,W; Lane==1 is the second bit
19493 // set, etc.
19494 Lane = SubIdx2Lane(User->getConstantOperandVal(1));
19495 if (Lane == ~0u)
19496 return Node;
19497
19498 // Check if the use is for the TFE/LWE generated result at VGPRn+1.
19499 if (UsesTFC && Lane == TFCLane) {
19500 Users[Lane] = User;
19501 } else {
19502 // Set which texture component corresponds to the lane.
19503 unsigned Comp;
19504 for (unsigned i = 0, Dmask = OldDmask; (i <= Lane) && (Dmask != 0); i++) {
19505 Comp = llvm::countr_zero(Dmask);
19506 Dmask &= ~(1 << Comp);
19507 }
19508
19509 // Abort if we have more than one user per component.
19510 if (Users[Lane])
19511 return Node;
19512
19513 Users[Lane] = User;
19514 NewDmask |= 1 << Comp;
19515 }
19516 }
19517
19518 // Don't allow 0 dmask, as hardware assumes one channel enabled.
19519 bool NoChannels = !NewDmask;
19520 if (NoChannels) {
19521 if (!UsesTFC) {
19522 // No uses of the result and not using TFC. Then do nothing.
19523 return Node;
19524 }
19525 // If the original dmask has one channel - then nothing to do
19526 if (OldBitsSet == 1)
19527 return Node;
19528 // Use an arbitrary dmask - required for the instruction to work
19529 NewDmask = 1;
19530 }
19531 // Abort if there's no change
19532 if (NewDmask == OldDmask)
19533 return Node;
19534
19535 unsigned BitsSet = llvm::popcount(NewDmask);
19536
19537 // Check for TFE or LWE - increase the number of channels by one to account
19538 // for the extra return value
19539 // This will need adjustment for D16 if this is also included in
19540 // adjustWriteMask (this function) but at present D16 are excluded.
19541 unsigned NewChannels = BitsSet + UsesTFC;
19542
19543 int NewOpcode =
19544 AMDGPU::getMaskedMIMGOp(Node->getMachineOpcode(), NewChannels);
19545 assert(NewOpcode != -1 &&
19546 NewOpcode != static_cast<int>(Node->getMachineOpcode()) &&
19547 "failed to find equivalent MIMG op");
19548
19549 // Adjust the writemask in the node
19551 llvm::append_range(Ops, Node->ops().take_front(DmaskIdx));
19552 Ops.push_back(DAG.getTargetConstant(NewDmask, SDLoc(Node), MVT::i32));
19553 llvm::append_range(Ops, Node->ops().drop_front(DmaskIdx + 1));
19554
19555 MVT SVT = Node->getValueType(0).getVectorElementType().getSimpleVT();
19556
19557 MVT ResultVT = NewChannels == 1
19558 ? SVT
19559 : MVT::getVectorVT(SVT, NewChannels == 3 ? 4
19560 : NewChannels == 5 ? 8
19561 : NewChannels);
19562 SDVTList NewVTList =
19563 HasChain ? DAG.getVTList(ResultVT, MVT::Other) : DAG.getVTList(ResultVT);
19564
19565 MachineSDNode *NewNode =
19566 DAG.getMachineNode(NewOpcode, SDLoc(Node), NewVTList, Ops);
19567
19568 if (HasChain) {
19569 // Update chain.
19570 DAG.setNodeMemRefs(NewNode, Node->memoperands());
19571 DAG.ReplaceAllUsesOfValueWith(SDValue(Node, 1), SDValue(NewNode, 1));
19572 }
19573
19574 if (NewChannels == 1) {
19575 assert(Node->hasNUsesOfValue(1, 0));
19576 SDNode *Copy =
19577 DAG.getMachineNode(TargetOpcode::COPY, SDLoc(Node),
19578 Users[Lane]->getValueType(0), SDValue(NewNode, 0));
19579 DAG.ReplaceAllUsesWith(Users[Lane], Copy);
19580 return nullptr;
19581 }
19582
19583 // Update the users of the node with the new indices
19584 for (unsigned i = 0, Idx = AMDGPU::sub0; i < 5; ++i) {
19585 SDNode *User = Users[i];
19586 if (!User) {
19587 // Handle the special case of NoChannels. We set NewDmask to 1 above, but
19588 // Users[0] is still nullptr because channel 0 doesn't really have a use.
19589 if (i || !NoChannels)
19590 continue;
19591 } else {
19592 SDValue Op = DAG.getTargetConstant(Idx, SDLoc(User), MVT::i32);
19593 SDNode *NewUser = DAG.UpdateNodeOperands(User, SDValue(NewNode, 0), Op);
19594 if (NewUser != User) {
19595 DAG.ReplaceAllUsesWith(SDValue(User, 0), SDValue(NewUser, 0));
19596 DAG.RemoveDeadNode(User);
19597 }
19598 }
19599
19600 switch (Idx) {
19601 default:
19602 break;
19603 case AMDGPU::sub0:
19604 Idx = AMDGPU::sub1;
19605 break;
19606 case AMDGPU::sub1:
19607 Idx = AMDGPU::sub2;
19608 break;
19609 case AMDGPU::sub2:
19610 Idx = AMDGPU::sub3;
19611 break;
19612 case AMDGPU::sub3:
19613 Idx = AMDGPU::sub4;
19614 break;
19615 }
19616 }
19617
19618 DAG.RemoveDeadNode(Node);
19619 return nullptr;
19620}
19621
19623 if (Op.getOpcode() == ISD::AssertZext)
19624 Op = Op.getOperand(0);
19625
19626 return isa<FrameIndexSDNode>(Op);
19627}
19628
19629/// Legalize target independent instructions (e.g. INSERT_SUBREG)
19630/// with frame index operands.
19631/// LLVM assumes that inputs are to these instructions are registers.
19632SDNode *
19634 SelectionDAG &DAG) const {
19635 if (Node->getOpcode() == ISD::CopyToReg) {
19636 RegisterSDNode *DestReg = cast<RegisterSDNode>(Node->getOperand(1));
19637 SDValue SrcVal = Node->getOperand(2);
19638
19639 // Insert a copy to a VReg_1 virtual register so LowerI1Copies doesn't have
19640 // to try understanding copies to physical registers.
19641 if (SrcVal.getValueType() == MVT::i1 && DestReg->getReg().isPhysical()) {
19642 SDLoc SL(Node);
19644 SDValue VReg = DAG.getRegister(
19645 MRI.createVirtualRegister(&AMDGPU::VReg_1RegClass), MVT::i1);
19646
19647 SDNode *Glued = Node->getGluedNode();
19648 SDValue ToVReg = DAG.getCopyToReg(
19649 Node->getOperand(0), SL, VReg, SrcVal,
19650 SDValue(Glued, Glued ? Glued->getNumValues() - 1 : 0));
19651 SDValue ToResultReg = DAG.getCopyToReg(ToVReg, SL, SDValue(DestReg, 0),
19652 VReg, ToVReg.getValue(1));
19653 DAG.ReplaceAllUsesWith(Node, ToResultReg.getNode());
19654 DAG.RemoveDeadNode(Node);
19655 return ToResultReg.getNode();
19656 }
19657 }
19658
19660 for (unsigned i = 0; i < Node->getNumOperands(); ++i) {
19661 if (!isFrameIndexOp(Node->getOperand(i))) {
19662 Ops.push_back(Node->getOperand(i));
19663 continue;
19664 }
19665
19666 SDLoc DL(Node);
19667 Ops.push_back(SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B32, DL,
19668 Node->getOperand(i).getValueType(),
19669 Node->getOperand(i)),
19670 0));
19671 }
19672
19673 return DAG.UpdateNodeOperands(Node, Ops);
19674}
19675
19676/// Fold the instructions after selecting them.
19677/// Returns null if users were already updated.
19679 SelectionDAG &DAG) const {
19681 unsigned Opcode = Node->getMachineOpcode();
19682
19683 if (TII->isImage(Opcode) && !TII->get(Opcode).mayStore() &&
19684 !TII->isGather4(Opcode) &&
19685 AMDGPU::hasNamedOperand(Opcode, AMDGPU::OpName::dmask)) {
19686 return adjustWritemask(Node, DAG);
19687 }
19688
19689 if (Opcode == AMDGPU::INSERT_SUBREG || Opcode == AMDGPU::REG_SEQUENCE) {
19691 return Node;
19692 }
19693
19694 switch (Opcode) {
19695 case AMDGPU::V_DIV_SCALE_F32_e64:
19696 case AMDGPU::V_DIV_SCALE_F64_e64: {
19697 // Satisfy the operand register constraint when one of the inputs is
19698 // undefined. Ordinarily each undef value will have its own implicit_def of
19699 // a vreg, so force these to use a single register.
19700 SDValue Src0 = Node->getOperand(1);
19701 SDValue Src1 = Node->getOperand(3);
19702 SDValue Src2 = Node->getOperand(5);
19703
19704 if ((Src0.isMachineOpcode() &&
19705 Src0.getMachineOpcode() != AMDGPU::IMPLICIT_DEF) &&
19706 (Src0 == Src1 || Src0 == Src2))
19707 break;
19708
19709 MVT VT = Src0.getValueType().getSimpleVT();
19710 const TargetRegisterClass *RC =
19711 getRegClassFor(VT, Src0.getNode()->isDivergent());
19712
19714 SDValue UndefReg = DAG.getRegister(MRI.createVirtualRegister(RC), VT);
19715
19716 SDValue ImpDef = DAG.getCopyToReg(DAG.getEntryNode(), SDLoc(Node), UndefReg,
19717 Src0, SDValue());
19718
19719 // src0 must be the same register as src1 or src2, even if the value is
19720 // undefined, so make sure we don't violate this constraint.
19721 if (Src0.isMachineOpcode() &&
19722 Src0.getMachineOpcode() == AMDGPU::IMPLICIT_DEF) {
19723 if (Src1.isMachineOpcode() &&
19724 Src1.getMachineOpcode() != AMDGPU::IMPLICIT_DEF)
19725 Src0 = Src1;
19726 else if (Src2.isMachineOpcode() &&
19727 Src2.getMachineOpcode() != AMDGPU::IMPLICIT_DEF)
19728 Src0 = Src2;
19729 else {
19730 assert(Src1.getMachineOpcode() == AMDGPU::IMPLICIT_DEF);
19731 Src0 = UndefReg;
19732 Src1 = UndefReg;
19733 }
19734 } else
19735 break;
19736
19738 Ops[1] = Src0;
19739 Ops[3] = Src1;
19740 Ops[5] = Src2;
19741 Ops.push_back(ImpDef.getValue(1));
19742 return DAG.getMachineNode(Opcode, SDLoc(Node), Node->getVTList(), Ops);
19743 }
19744 default:
19745 break;
19746 }
19747
19748 return Node;
19749}
19750
19751// Any MIMG instructions that use tfe or lwe require an initialization of the
19752// result register that will be written in the case of a memory access failure.
19753// The required code is also added to tie this init code to the result of the
19754// img instruction.
19757 const SIRegisterInfo &TRI = TII->getRegisterInfo();
19758 MachineRegisterInfo &MRI = MI.getMF()->getRegInfo();
19759 MachineBasicBlock &MBB = *MI.getParent();
19760
19761 int DstIdx =
19762 AMDGPU::getNamedOperandIdx(MI.getOpcode(), AMDGPU::OpName::vdata);
19763 unsigned InitIdx = 0;
19764
19765 if (TII->isImage(MI)) {
19766 MachineOperand *TFE = TII->getNamedOperand(MI, AMDGPU::OpName::tfe);
19767 MachineOperand *LWE = TII->getNamedOperand(MI, AMDGPU::OpName::lwe);
19768 MachineOperand *D16 = TII->getNamedOperand(MI, AMDGPU::OpName::d16);
19769
19770 if (!TFE && !LWE) // intersect_ray
19771 return;
19772
19773 unsigned TFEVal = TFE ? TFE->getImm() : 0;
19774 unsigned LWEVal = LWE ? LWE->getImm() : 0;
19775 unsigned D16Val = D16 ? D16->getImm() : 0;
19776
19777 if (!TFEVal && !LWEVal)
19778 return;
19779
19780 // At least one of TFE or LWE are non-zero
19781 // We have to insert a suitable initialization of the result value and
19782 // tie this to the dest of the image instruction.
19783
19784 // Calculate which dword we have to initialize to 0.
19785 MachineOperand *MO_Dmask = TII->getNamedOperand(MI, AMDGPU::OpName::dmask);
19786
19787 // check that dmask operand is found.
19788 assert(MO_Dmask && "Expected dmask operand in instruction");
19789
19790 unsigned dmask = MO_Dmask->getImm();
19791 // Determine the number of active lanes taking into account the
19792 // Gather4 special case
19793 unsigned ActiveLanes = TII->isGather4(MI) ? 4 : llvm::popcount(dmask);
19794
19795 bool Packed = !Subtarget->hasUnpackedD16VMem();
19796
19797 InitIdx = D16Val && Packed ? ((ActiveLanes + 1) >> 1) + 1 : ActiveLanes + 1;
19798
19799 // Abandon attempt if the dst size isn't large enough
19800 // - this is in fact an error but this is picked up elsewhere and
19801 // reported correctly.
19802 const TargetRegisterClass *DstRC = TII->getRegClass(MI.getDesc(), DstIdx);
19803
19804 uint32_t DstSize = TRI.getRegSizeInBits(*DstRC) / 32;
19805 if (DstSize < InitIdx)
19806 return;
19807 } else if (TII->isMUBUF(MI) && AMDGPU::getMUBUFTfe(MI.getOpcode())) {
19808 const TargetRegisterClass *DstRC = TII->getRegClass(MI.getDesc(), DstIdx);
19809 InitIdx = TRI.getRegSizeInBits(*DstRC) / 32;
19810 } else {
19811 return;
19812 }
19813
19814 const DebugLoc &DL = MI.getDebugLoc();
19815
19816 // Create a register for the initialization value.
19817 Register PrevDst = MRI.cloneVirtualRegister(MI.getOperand(DstIdx).getReg());
19818 unsigned NewDst = 0; // Final initialized value will be in here
19819
19820 // If PRTStrictNull feature is enabled (the default) then initialize
19821 // all the result registers to 0, otherwise just the error indication
19822 // register (VGPRn+1)
19823 unsigned SizeLeft = Subtarget->usePRTStrictNull() ? InitIdx : 1;
19824 unsigned CurrIdx = Subtarget->usePRTStrictNull() ? 0 : (InitIdx - 1);
19825
19826 BuildMI(MBB, MI, DL, TII->get(AMDGPU::IMPLICIT_DEF), PrevDst);
19827 for (; SizeLeft; SizeLeft--, CurrIdx++) {
19828 NewDst = MRI.createVirtualRegister(TII->getOpRegClass(MI, DstIdx));
19829 // Initialize dword
19830 Register SubReg = MRI.createVirtualRegister(&AMDGPU::VGPR_32RegClass);
19831 // clang-format off
19832 BuildMI(MBB, MI, DL, TII->get(AMDGPU::V_MOV_B32_e32), SubReg)
19833 .addImm(0);
19834 // clang-format on
19835 // Insert into the super-reg
19836 BuildMI(MBB, MI, DL, TII->get(TargetOpcode::INSERT_SUBREG), NewDst)
19837 .addReg(PrevDst)
19838 .addReg(SubReg)
19840
19841 PrevDst = NewDst;
19842 }
19843
19844 // Add as an implicit operand
19845 MI.addOperand(MachineOperand::CreateReg(NewDst, false, true));
19846
19847 // Tie the just added implicit operand to the dst
19848 MI.tieOperands(DstIdx, MI.getNumOperands() - 1);
19849}
19850
19851/// Assign the register class depending on the number of
19852/// bits set in the writemask
19854 SDNode *Node) const {
19856
19857 MachineFunction *MF = MI.getMF();
19858 MachineRegisterInfo &MRI = MF->getRegInfo();
19859
19860 if (TII->isVOP3(MI.getOpcode())) {
19861 // Make sure constant bus requirements are respected.
19862 TII->legalizeOperandsVOP3(MRI, MI);
19863
19864 if (TII->isMAI(MI)) {
19865 // The ordinary src0, src1, src2 were legalized above.
19866 //
19867 // We have to also legalize the appended v_mfma_ld_scale_b32 operands,
19868 // as a separate instruction.
19869 int Src0Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
19870 AMDGPU::OpName::scale_src0);
19871 if (Src0Idx != -1) {
19872 int Src1Idx = AMDGPU::getNamedOperandIdx(MI.getOpcode(),
19873 AMDGPU::OpName::scale_src1);
19874 if (TII->usesConstantBus(MRI, MI, Src0Idx) &&
19875 TII->usesConstantBus(MRI, MI, Src1Idx))
19876 TII->legalizeOpWithMove(MI, Src1Idx);
19877 }
19878 }
19879
19880 return;
19881 }
19882
19883 if (TII->isImage(MI))
19884 TII->enforceOperandRCAlignment(MI, AMDGPU::OpName::vaddr);
19885}
19886
19888 uint64_t Val) {
19889 SDValue K = DAG.getTargetConstant(Val, DL, MVT::i32);
19890 return SDValue(DAG.getMachineNode(AMDGPU::S_MOV_B32, DL, MVT::i32, K), 0);
19891}
19892
19894 const SDLoc &DL,
19895 SDValue Ptr) const {
19897
19898 // Build the half of the subregister with the constants before building the
19899 // full 128-bit register. If we are building multiple resource descriptors,
19900 // this will allow CSEing of the 2-component register.
19901 const SDValue Ops0[] = {
19902 DAG.getTargetConstant(AMDGPU::SGPR_64RegClassID, DL, MVT::i32),
19903 buildSMovImm32(DAG, DL, 0),
19904 DAG.getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
19905 buildSMovImm32(DAG, DL, TII->getDefaultRsrcDataFormat() >> 32),
19906 DAG.getTargetConstant(AMDGPU::sub1, DL, MVT::i32)};
19907
19908 SDValue SubRegHi = SDValue(
19909 DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL, MVT::v2i32, Ops0), 0);
19910
19911 // Combine the constants and the pointer.
19912 const SDValue Ops1[] = {
19913 DAG.getTargetConstant(AMDGPU::SGPR_128RegClassID, DL, MVT::i32), Ptr,
19914 DAG.getTargetConstant(AMDGPU::sub0_sub1, DL, MVT::i32), SubRegHi,
19915 DAG.getTargetConstant(AMDGPU::sub2_sub3, DL, MVT::i32)};
19916
19917 return DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL, MVT::v4i32, Ops1);
19918}
19919
19920/// Return a resource descriptor with the 'Add TID' bit enabled
19921/// The TID (Thread ID) is multiplied by the stride value (bits [61:48]
19922/// of the resource descriptor) to create an offset, which is added to
19923/// the resource pointer.
19925 SDValue Ptr, uint32_t RsrcDword1,
19926 uint64_t RsrcDword2And3) const {
19927 SDValue PtrLo = DAG.getTargetExtractSubreg(AMDGPU::sub0, DL, MVT::i32, Ptr);
19928 SDValue PtrHi = DAG.getTargetExtractSubreg(AMDGPU::sub1, DL, MVT::i32, Ptr);
19929 if (RsrcDword1) {
19930 PtrHi = DAG.getNode(ISD::OR, DL, MVT::i32, PtrHi,
19931 DAG.getConstant(RsrcDword1, DL, MVT::i32));
19932 }
19933
19934 SDValue DataLo =
19935 buildSMovImm32(DAG, DL, RsrcDword2And3 & UINT64_C(0xFFFFFFFF));
19936 SDValue DataHi = buildSMovImm32(DAG, DL, RsrcDword2And3 >> 32);
19937
19938 const SDValue Ops[] = {
19939 DAG.getTargetConstant(AMDGPU::SGPR_128RegClassID, DL, MVT::i32),
19940 PtrLo,
19941 DAG.getTargetConstant(AMDGPU::sub0, DL, MVT::i32),
19942 PtrHi,
19943 DAG.getTargetConstant(AMDGPU::sub1, DL, MVT::i32),
19944 DataLo,
19945 DAG.getTargetConstant(AMDGPU::sub2, DL, MVT::i32),
19946 DataHi,
19947 DAG.getTargetConstant(AMDGPU::sub3, DL, MVT::i32)};
19948
19949 return DAG.getMachineNode(AMDGPU::REG_SEQUENCE, DL, MVT::v4i32, Ops);
19950}
19951
19952//===----------------------------------------------------------------------===//
19953// SI Inline Assembly Support
19954//===----------------------------------------------------------------------===//
19955
19956std::pair<unsigned, const TargetRegisterClass *>
19958 StringRef Constraint,
19959 MVT VT) const {
19960 const SIRegisterInfo *TRI = static_cast<const SIRegisterInfo *>(TRI_);
19961
19962 const TargetRegisterClass *RC = nullptr;
19963 if (Constraint.size() == 1) {
19964 // Check if we cannot determine the bit size of the given value type. This
19965 // can happen, for example, in this situation where we have an empty struct
19966 // (size 0): `call void asm "", "v"({} poison)`-
19967 if (VT == MVT::Other)
19968 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
19969 const unsigned BitWidth = VT.getSizeInBits();
19970 switch (Constraint[0]) {
19971 default:
19972 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
19973 case 's':
19974 case 'r':
19975 switch (BitWidth) {
19976 case 16:
19977 RC = &AMDGPU::SReg_32RegClass;
19978 break;
19979 case 64:
19980 RC = &AMDGPU::SGPR_64RegClass;
19981 break;
19982 default:
19984 if (!RC)
19985 return std::pair(0U, nullptr);
19986 break;
19987 }
19988 break;
19989 case 'v':
19990 switch (BitWidth) {
19991 case 1:
19992 return std::pair(0U, nullptr);
19993 case 16:
19994 RC = Subtarget->useRealTrue16Insts() ? &AMDGPU::VGPR_16RegClass
19995 : &AMDGPU::VGPR_32_Lo256RegClass;
19996 break;
19997 default:
19998 RC = Subtarget->has1024AddressableVGPRs()
19999 ? TRI->getAlignedLo256VGPRClassForBitWidth(BitWidth)
20000 : TRI->getVGPRClassForBitWidth(BitWidth);
20001 if (!RC)
20002 return std::pair(0U, nullptr);
20003 break;
20004 }
20005 break;
20006 case 'a':
20007 if (!Subtarget->hasMAIInsts())
20008 break;
20009 switch (BitWidth) {
20010 case 1:
20011 return std::pair(0U, nullptr);
20012 case 16:
20013 RC = &AMDGPU::AGPR_32RegClass;
20014 break;
20015 default:
20016 RC = TRI->getAGPRClassForBitWidth(BitWidth);
20017 if (!RC)
20018 return std::pair(0U, nullptr);
20019 break;
20020 }
20021 break;
20022 }
20023 } else if (Constraint == "VA" && Subtarget->hasGFX90AInsts()) {
20024 const unsigned BitWidth = VT.getSizeInBits();
20025 switch (BitWidth) {
20026 case 16:
20027 RC = &AMDGPU::AV_32RegClass;
20028 break;
20029 default:
20030 RC = TRI->getVectorSuperClassForBitWidth(BitWidth);
20031 if (!RC)
20032 return std::pair(0U, nullptr);
20033 break;
20034 }
20035 }
20036
20037 // We actually support i128, i16 and f16 as inline parameters
20038 // even if they are not reported as legal
20039 if (RC && (isTypeLegal(VT) || VT.SimpleTy == MVT::i128 ||
20040 VT.SimpleTy == MVT::i16 || VT.SimpleTy == MVT::f16))
20041 return std::pair(0U, RC);
20042
20043 auto [Kind, Idx, NumRegs] = AMDGPU::parseAsmConstraintPhysReg(Constraint);
20044 if (Kind != '\0') {
20045 if (Kind == 'v') {
20046 RC = &AMDGPU::VGPR_32_Lo256RegClass;
20047 } else if (Kind == 's') {
20048 RC = &AMDGPU::SGPR_32RegClass;
20049 } else if (Kind == 'a') {
20050 RC = &AMDGPU::AGPR_32RegClass;
20051 }
20052
20053 if (RC) {
20054 if (NumRegs > 1) {
20055 if (Idx >= RC->getNumRegs() || Idx + NumRegs - 1 >= RC->getNumRegs())
20056 return std::pair(0U, nullptr);
20057
20058 uint32_t Width = NumRegs * 32;
20059 // Prohibit constraints for register ranges with a width that does not
20060 // match the required type.
20061 if (VT.SimpleTy != MVT::Other && Width != VT.getSizeInBits())
20062 return std::pair(0U, nullptr);
20063
20064 MCRegister Reg = RC->getRegister(Idx);
20066 RC = TRI->getVGPRClassForBitWidth(Width);
20067 else if (SIRegisterInfo::isSGPRClass(RC))
20068 RC = TRI->getSGPRClassForBitWidth(Width);
20069 else if (SIRegisterInfo::isAGPRClass(RC))
20070 RC = TRI->getAGPRClassForBitWidth(Width);
20071 if (RC) {
20072 Reg = TRI->getMatchingSuperReg(Reg, AMDGPU::sub0, RC);
20073 if (!Reg) {
20074 // The register class does not contain the requested register,
20075 // e.g., because it is an SGPR pair that would violate alignment
20076 // requirements.
20077 return std::pair(0U, nullptr);
20078 }
20079 return std::pair(Reg, RC);
20080 }
20081 }
20082
20083 // Reject types that do not fit a single 32-bit register: any scalar wider
20084 // than 32 bits, or a vector that is not exactly 32 bits.
20085 if (VT.SimpleTy != MVT::Other &&
20086 (VT.getSizeInBits() > 32 ||
20087 (VT.isVector() && VT.getSizeInBits() != 32)))
20088 return std::pair(0U, nullptr);
20089 if (RC && Idx < RC->getNumRegs())
20090 return std::pair(RC->getRegister(Idx), RC);
20091 return std::pair(0U, nullptr);
20092 }
20093 }
20094
20095 auto Ret = TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
20096 if (Ret.first)
20097 Ret.second = TRI->getPhysRegBaseClass(Ret.first);
20098
20099 return Ret;
20100}
20101
20102static bool isImmConstraint(StringRef Constraint) {
20103 if (Constraint.size() == 1) {
20104 switch (Constraint[0]) {
20105 default:
20106 break;
20107 case 'I':
20108 case 'J':
20109 case 'A':
20110 case 'B':
20111 case 'C':
20112 return true;
20113 }
20114 } else if (Constraint == "DA" || Constraint == "DB") {
20115 return true;
20116 }
20117 return false;
20118}
20119
20122 if (Constraint.size() == 1) {
20123 switch (Constraint[0]) {
20124 default:
20125 break;
20126 case 's':
20127 case 'v':
20128 case 'a':
20129 return C_RegisterClass;
20130 }
20131 } else if (Constraint.size() == 2) {
20132 if (Constraint == "VA")
20133 return C_RegisterClass;
20134 }
20135 if (isImmConstraint(Constraint)) {
20136 return C_Other;
20137 }
20138 return TargetLowering::getConstraintType(Constraint);
20139}
20140
20141static uint64_t clearUnusedBits(uint64_t Val, unsigned Size) {
20143 Val = Val & maskTrailingOnes<uint64_t>(Size);
20144 }
20145 return Val;
20146}
20147
20149 StringRef Constraint,
20150 std::vector<SDValue> &Ops,
20151 SelectionDAG &DAG) const {
20152 if (isImmConstraint(Constraint)) {
20153 uint64_t Val;
20154 if (getAsmOperandConstVal(Op, Val) &&
20155 checkAsmConstraintVal(Op, Constraint, Val)) {
20156 Val = clearUnusedBits(Val, Op.getScalarValueSizeInBits());
20157 Ops.push_back(DAG.getTargetConstant(Val, SDLoc(Op), MVT::i64));
20158 }
20159 } else {
20161 }
20162}
20163
20165 unsigned Size = Op.getScalarValueSizeInBits();
20166 if (Size > 64)
20167 return false;
20168
20169 if (Size == 16 && !Subtarget->has16BitInsts())
20170 return false;
20171
20173 Val = C->getSExtValue();
20174 return true;
20175 }
20177 Val = C->getValueAPF().bitcastToAPInt().getSExtValue();
20178 return true;
20179 }
20181 if (Size != 16 || Op.getNumOperands() != 2)
20182 return false;
20183 if (Op.getOperand(0).isUndef() || Op.getOperand(1).isUndef())
20184 return false;
20185 if (ConstantSDNode *C = V->getConstantSplatNode()) {
20186 Val = C->getSExtValue();
20187 return true;
20188 }
20189 if (ConstantFPSDNode *C = V->getConstantFPSplatNode()) {
20190 Val = C->getValueAPF().bitcastToAPInt().getSExtValue();
20191 return true;
20192 }
20193 }
20194
20195 return false;
20196}
20197
20199 uint64_t Val) const {
20200 if (Constraint.size() == 1) {
20201 switch (Constraint[0]) {
20202 case 'I':
20204 case 'J':
20205 return isInt<16>(Val);
20206 case 'A':
20207 return checkAsmConstraintValA(Op, Val);
20208 case 'B':
20209 return isInt<32>(Val);
20210 case 'C':
20211 return isUInt<32>(clearUnusedBits(Val, Op.getScalarValueSizeInBits())) ||
20213 default:
20214 break;
20215 }
20216 } else if (Constraint.size() == 2) {
20217 if (Constraint == "DA") {
20218 int64_t HiBits = static_cast<int32_t>(Val >> 32);
20219 int64_t LoBits = static_cast<int32_t>(Val);
20220 return checkAsmConstraintValA(Op, HiBits, 32) &&
20221 checkAsmConstraintValA(Op, LoBits, 32);
20222 }
20223 if (Constraint == "DB") {
20224 return true;
20225 }
20226 }
20227 llvm_unreachable("Invalid asm constraint");
20228}
20229
20231 unsigned MaxSize) const {
20232 unsigned Size = std::min<unsigned>(Op.getScalarValueSizeInBits(), MaxSize);
20233 bool HasInv2Pi = Subtarget->hasInv2PiInlineImm();
20234 if (Size == 16) {
20235 MVT VT = Op.getSimpleValueType();
20236 switch (VT.SimpleTy) {
20237 default:
20238 return false;
20239 case MVT::i16:
20240 return AMDGPU::isInlinableLiteralI16(Val, HasInv2Pi);
20241 case MVT::f16:
20242 return AMDGPU::isInlinableLiteralFP16(Val, HasInv2Pi);
20243 case MVT::bf16:
20244 return AMDGPU::isInlinableLiteralBF16(Val, HasInv2Pi);
20245 case MVT::v2i16:
20246 return AMDGPU::getInlineEncodingV2I16(Val).has_value();
20247 case MVT::v2f16:
20248 return AMDGPU::getInlineEncodingV2F16(Val).has_value();
20249 case MVT::v2bf16:
20250 return AMDGPU::getInlineEncodingV2BF16(Val).has_value();
20251 }
20252 }
20253 if ((Size == 32 && AMDGPU::isInlinableLiteral32(Val, HasInv2Pi)) ||
20254 (Size == 64 && AMDGPU::isInlinableLiteral64(Val, HasInv2Pi)))
20255 return true;
20256 return false;
20257}
20258
20259static int getAlignedAGPRClassID(unsigned UnalignedClassID) {
20260 switch (UnalignedClassID) {
20261 case AMDGPU::VReg_64RegClassID:
20262 return AMDGPU::VReg_64_Align2RegClassID;
20263 case AMDGPU::VReg_96RegClassID:
20264 return AMDGPU::VReg_96_Align2RegClassID;
20265 case AMDGPU::VReg_128RegClassID:
20266 return AMDGPU::VReg_128_Align2RegClassID;
20267 case AMDGPU::VReg_160RegClassID:
20268 return AMDGPU::VReg_160_Align2RegClassID;
20269 case AMDGPU::VReg_192RegClassID:
20270 return AMDGPU::VReg_192_Align2RegClassID;
20271 case AMDGPU::VReg_224RegClassID:
20272 return AMDGPU::VReg_224_Align2RegClassID;
20273 case AMDGPU::VReg_256RegClassID:
20274 return AMDGPU::VReg_256_Align2RegClassID;
20275 case AMDGPU::VReg_288RegClassID:
20276 return AMDGPU::VReg_288_Align2RegClassID;
20277 case AMDGPU::VReg_320RegClassID:
20278 return AMDGPU::VReg_320_Align2RegClassID;
20279 case AMDGPU::VReg_352RegClassID:
20280 return AMDGPU::VReg_352_Align2RegClassID;
20281 case AMDGPU::VReg_384RegClassID:
20282 return AMDGPU::VReg_384_Align2RegClassID;
20283 case AMDGPU::VReg_512RegClassID:
20284 return AMDGPU::VReg_512_Align2RegClassID;
20285 case AMDGPU::VReg_1024RegClassID:
20286 return AMDGPU::VReg_1024_Align2RegClassID;
20287 case AMDGPU::AReg_64RegClassID:
20288 return AMDGPU::AReg_64_Align2RegClassID;
20289 case AMDGPU::AReg_96RegClassID:
20290 return AMDGPU::AReg_96_Align2RegClassID;
20291 case AMDGPU::AReg_128RegClassID:
20292 return AMDGPU::AReg_128_Align2RegClassID;
20293 case AMDGPU::AReg_160RegClassID:
20294 return AMDGPU::AReg_160_Align2RegClassID;
20295 case AMDGPU::AReg_192RegClassID:
20296 return AMDGPU::AReg_192_Align2RegClassID;
20297 case AMDGPU::AReg_256RegClassID:
20298 return AMDGPU::AReg_256_Align2RegClassID;
20299 case AMDGPU::AReg_512RegClassID:
20300 return AMDGPU::AReg_512_Align2RegClassID;
20301 case AMDGPU::AReg_1024RegClassID:
20302 return AMDGPU::AReg_1024_Align2RegClassID;
20303 default:
20304 return -1;
20305 }
20306}
20307
20308// Figure out which registers should be reserved for stack access. Only after
20309// the function is legalized do we know all of the non-spill stack objects or if
20310// calls are present.
20312 MachineRegisterInfo &MRI = MF.getRegInfo();
20314 const GCNSubtarget &ST = MF.getSubtarget<GCNSubtarget>();
20315 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
20316 const SIInstrInfo *TII = ST.getInstrInfo();
20317
20318 if (Info->isEntryFunction()) {
20319 // Callable functions have fixed registers used for stack access.
20321 }
20322
20323 // TODO: Move this logic to getReservedRegs()
20324 // Reserve the SGPR(s) to save/restore EXEC for WWM spill/copy handling.
20325 unsigned MaxNumSGPRs = ST.getMaxNumSGPRs(MF);
20326 Register SReg = ST.isWave32()
20327 ? AMDGPU::SGPR_32RegClass.getRegister(MaxNumSGPRs - 1)
20328 : TRI->getAlignedHighSGPRForRC(MF, /*Align=*/2,
20329 &AMDGPU::SGPR_64RegClass);
20330 Info->setSGPRForEXECCopy(SReg);
20331
20332 assert(!TRI->isSubRegister(Info->getScratchRSrcReg(),
20333 Info->getStackPtrOffsetReg()));
20334 if (Info->getStackPtrOffsetReg() != AMDGPU::SP_REG)
20335 MRI.replaceRegWith(AMDGPU::SP_REG, Info->getStackPtrOffsetReg());
20336
20337 // We need to worry about replacing the default register with itself in case
20338 // of MIR testcases missing the MFI.
20339 if (Info->getScratchRSrcReg() != AMDGPU::PRIVATE_RSRC_REG)
20340 MRI.replaceRegWith(AMDGPU::PRIVATE_RSRC_REG, Info->getScratchRSrcReg());
20341
20342 if (Info->getFrameOffsetReg() != AMDGPU::FP_REG)
20343 MRI.replaceRegWith(AMDGPU::FP_REG, Info->getFrameOffsetReg());
20344
20345 Info->limitOccupancy(MF);
20346
20347 if (ST.isWave32() && !MF.empty()) {
20348 for (auto &MBB : MF) {
20349 for (auto &MI : MBB) {
20350 TII->fixImplicitOperands(MI);
20351 }
20352 }
20353 }
20354
20355 // FIXME: This is a hack to fixup AGPR classes to use the properly aligned
20356 // classes if required. Ideally the register class constraints would differ
20357 // per-subtarget, but there's no easy way to achieve that right now. This is
20358 // not a problem for VGPRs because the correctly aligned VGPR class is implied
20359 // from using them as the register class for legal types.
20360 if (ST.needsAlignedVGPRs()) {
20361 for (unsigned I = 0, E = MRI.getNumVirtRegs(); I != E; ++I) {
20362 const Register Reg = Register::index2VirtReg(I);
20363 const TargetRegisterClass *RC = MRI.getRegClassOrNull(Reg);
20364 if (!RC)
20365 continue;
20366 int NewClassID = getAlignedAGPRClassID(RC->getID());
20367 if (NewClassID != -1)
20368 MRI.setRegClass(Reg, TRI->getRegClass(NewClassID));
20369 }
20370 }
20371
20373}
20374
20377 const APInt &DemandedElts,
20378 const SelectionDAG &DAG,
20379 unsigned Depth) const {
20380 Known.resetAll();
20381 unsigned Opc = Op.getOpcode();
20382 switch (Opc) {
20384 unsigned IID = Op.getConstantOperandVal(0);
20385 switch (IID) {
20386 case Intrinsic::amdgcn_mbcnt_lo:
20387 case Intrinsic::amdgcn_mbcnt_hi: {
20388 const GCNSubtarget &ST =
20390 // Wave64 mbcnt_lo returns at most 32 + src1. Otherwise these return at
20391 // most 31 + src1.
20392 Known.Zero.setBitsFrom(
20393 IID == Intrinsic::amdgcn_mbcnt_lo ? ST.getWavefrontSizeLog2() : 5);
20394 KnownBits Known2 = DAG.computeKnownBits(Op.getOperand(2), Depth + 1);
20395 Known = KnownBits::add(Known, Known2);
20396 return;
20397 }
20398 }
20399 break;
20400 }
20401 }
20403 Op, Known, DemandedElts, DAG, Depth);
20404}
20405
20407 KnownBits &Known, const MachineFunction &MF, Align Alignment) const {
20409
20410 // Set the high bits to zero based on the maximum allowed scratch size per
20411 // wave. We can't use vaddr in MUBUF instructions if we don't know the address
20412 // calculation won't overflow, so assume the sign bit is never set.
20413 Known.Zero.setHighBits(getSubtarget()->getKnownHighZeroBitsForFrameIndex());
20414}
20415
20418 unsigned Dim) {
20419 unsigned MaxValue =
20420 ST.getMaxWorkitemID(VT.getMachineFunction().getFunction(), Dim);
20421 Known.Zero.setHighBits(llvm::countl_zero(MaxValue));
20422}
20423
20425 KnownBits &Known, const APInt &DemandedElts,
20426 unsigned BFEWidth, bool SExt, unsigned Depth) {
20428 const MachineOperand &Src1 = MI.getOperand(2);
20429
20430 unsigned Src1Cst = 0;
20431 if (Src1.isImm()) {
20432 Src1Cst = Src1.getImm();
20433 } else if (Src1.isReg()) {
20434 auto Cst = getIConstantVRegValWithLookThrough(Src1.getReg(), MRI);
20435 if (!Cst)
20436 return;
20437 Src1Cst = Cst->Value.getZExtValue();
20438 } else {
20439 return;
20440 }
20441
20442 // Offset is at bits [4:0] for 32 bit, [5:0] for 64 bit.
20443 // Width is always [22:16].
20444 const unsigned Offset =
20445 Src1Cst & maskTrailingOnes<unsigned>((BFEWidth == 32) ? 5 : 6);
20446 const unsigned Width = (Src1Cst >> 16) & maskTrailingOnes<unsigned>(6);
20447
20448 if (Width >= BFEWidth) // Ill-formed.
20449 return;
20450
20451 VT.computeKnownBitsImpl(MI.getOperand(1).getReg(), Known, DemandedElts,
20452 Depth + 1);
20453
20454 Known = Known.extractBits(Width, Offset);
20455
20456 if (SExt)
20457 Known = Known.sext(BFEWidth);
20458 else
20459 Known = Known.zext(BFEWidth);
20460}
20461
20464 const APInt &DemandedElts, const MachineRegisterInfo &MRI,
20465 unsigned Depth) const {
20466 Known.resetAll();
20467 const MachineInstr *MI = MRI.getVRegDef(R);
20468 switch (MI->getOpcode()) {
20469 case AMDGPU::S_BFE_I32:
20470 return knownBitsForSBFE(*MI, VT, Known, DemandedElts, /*Width=*/32,
20471 /*SExt=*/true, Depth);
20472 case AMDGPU::S_BFE_U32:
20473 return knownBitsForSBFE(*MI, VT, Known, DemandedElts, /*Width=*/32,
20474 /*SExt=*/false, Depth);
20475 case AMDGPU::S_BFE_I64:
20476 return knownBitsForSBFE(*MI, VT, Known, DemandedElts, /*Width=*/64,
20477 /*SExt=*/true, Depth);
20478 case AMDGPU::S_BFE_U64:
20479 return knownBitsForSBFE(*MI, VT, Known, DemandedElts, /*Width=*/64,
20480 /*SExt=*/false, Depth);
20481 case AMDGPU::G_INTRINSIC:
20482 case AMDGPU::G_INTRINSIC_CONVERGENT: {
20483 Intrinsic::ID IID = cast<GIntrinsic>(MI)->getIntrinsicID();
20484 switch (IID) {
20485 case Intrinsic::amdgcn_workitem_id_x:
20487 break;
20488 case Intrinsic::amdgcn_workitem_id_y:
20490 break;
20491 case Intrinsic::amdgcn_workitem_id_z:
20493 break;
20494 case Intrinsic::amdgcn_mbcnt_lo:
20495 case Intrinsic::amdgcn_mbcnt_hi: {
20496 // Wave64 mbcnt_lo returns at most 32 + src1. Otherwise these return at
20497 // most 31 + src1.
20498 Known.Zero.setBitsFrom(IID == Intrinsic::amdgcn_mbcnt_lo
20499 ? getSubtarget()->getWavefrontSizeLog2()
20500 : 5);
20501 KnownBits Known2;
20502 VT.computeKnownBitsImpl(MI->getOperand(3).getReg(), Known2, DemandedElts,
20503 Depth + 1);
20504 Known = KnownBits::add(Known, Known2);
20505 break;
20506 }
20507 case Intrinsic::amdgcn_groupstaticsize: {
20508 // We can report everything over the maximum size as 0. We can't report
20509 // based on the actual size because we don't know if it's accurate or not
20510 // at any given point.
20511 Known.Zero.setHighBits(
20512 llvm::countl_zero(getSubtarget()->getAddressableLocalMemorySize()));
20513 break;
20514 }
20515 case Intrinsic::amdgcn_readfirstlane:
20516 case Intrinsic::amdgcn_readlane: {
20517 // Result is the data operand's value from some lane.
20518 VT.computeKnownBitsImpl(MI->getOperand(2).getReg(), Known, DemandedElts,
20519 Depth + 1);
20520 break;
20521 }
20522 }
20523 break;
20524 }
20525 case AMDGPU::G_AMDGPU_BUFFER_LOAD_UBYTE:
20526 Known.Zero.setHighBits(24);
20527 break;
20528 case AMDGPU::G_AMDGPU_BUFFER_LOAD_USHORT:
20529 Known.Zero.setHighBits(16);
20530 break;
20531 case AMDGPU::G_AMDGPU_COPY_SCC_VCC:
20532 // G_AMDGPU_COPY_SCC_VCC converts a uniform boolean in VCC to SGPR s32,
20533 // producing exactly 0 or 1.
20534 Known.Zero.setHighBits(Known.getBitWidth() - 1);
20535 break;
20536 case AMDGPU::G_AMDGPU_SMED3:
20537 case AMDGPU::G_AMDGPU_UMED3: {
20538 auto [Dst, Src0, Src1, Src2] = MI->getFirst4Regs();
20539
20540 KnownBits Known2;
20541 VT.computeKnownBitsImpl(Src2, Known2, DemandedElts, Depth + 1);
20542 if (Known2.isUnknown())
20543 break;
20544
20545 KnownBits Known1;
20546 VT.computeKnownBitsImpl(Src1, Known1, DemandedElts, Depth + 1);
20547 if (Known1.isUnknown())
20548 break;
20549
20550 KnownBits Known0;
20551 VT.computeKnownBitsImpl(Src0, Known0, DemandedElts, Depth + 1);
20552 if (Known0.isUnknown())
20553 break;
20554
20555 // TODO: Handle LeadZero/LeadOne from UMIN/UMAX handling.
20556 Known.Zero = Known0.Zero & Known1.Zero & Known2.Zero;
20557 Known.One = Known0.One & Known1.One & Known2.One;
20558 break;
20559 }
20560 }
20561}
20562
20565 unsigned Depth) const {
20566 const MachineInstr *MI = MRI.getVRegDef(R);
20567 if (auto *GI = dyn_cast<GIntrinsic>(MI)) {
20568 // FIXME: Can this move to generic code? What about the case where the call
20569 // site specifies a lower alignment?
20570 Intrinsic::ID IID = GI->getIntrinsicID();
20572 AttributeList Attrs =
20573 Intrinsic::getAttributes(Ctx, IID, Intrinsic::getType(Ctx, IID));
20574 if (MaybeAlign RetAlign = Attrs.getRetAlignment())
20575 return *RetAlign;
20576 }
20577 return Align(1);
20578}
20579
20581 MachineLoop *ML, const MachineBasicBlock *BlockToAlign) const {
20583 const Align CacheLineAlign = Align(64);
20584
20585 // GFX950: Prevent an 8-byte instruction at the block being aligned from being
20586 // split by the 32-byte instruction fetch window boundary. This avoids a
20587 // significant fetch delay after a backward branch. We use 32-byte alignment
20588 // with max padding of 4 bytes (one s_nop), see
20589 // getMaxPermittedBytesForAlignment().
20590 if (ML && !DisableLoopAlignment &&
20591 getSubtarget()->hasLoopHeadInstSplitSensitivity()) {
20592 // Loop rotation can make the backedge destination a block other than the
20593 // LoopInfo header, so prefer the block the caller is actually aligning.
20594 if (!BlockToAlign)
20595 BlockToAlign = ML->getHeader();
20596 // Respect user-specified or previously set alignment.
20597 if (BlockToAlign->getAlignment() != PrefAlign)
20598 return BlockToAlign->getAlignment();
20599 if (needsFetchWindowAlignment(*BlockToAlign))
20600 return Align(32);
20601 }
20602
20603 // Pre-GFX10 target did not benefit from loop alignment
20604 if (!ML || DisableLoopAlignment || !getSubtarget()->hasInstPrefetch() ||
20605 getSubtarget()->hasInstFwdPrefetchBug())
20606 return PrefAlign;
20607
20608 // On GFX10 I$ is 4 x 64 bytes cache lines.
20609 // By default prefetcher keeps one cache line behind and reads two ahead.
20610 // We can modify it with S_INST_PREFETCH for larger loops to have two lines
20611 // behind and one ahead.
20612 // Therefor we can benefit from aligning loop headers if loop fits 192 bytes.
20613 // If loop fits 64 bytes it always spans no more than two cache lines and
20614 // does not need an alignment.
20615 // Else if loop is less or equal 128 bytes we do not need to modify prefetch,
20616 // Else if loop is less or equal 192 bytes we need two lines behind.
20617
20619 const MachineBasicBlock *Header = ML->getHeader();
20620 if (Header->getAlignment() != PrefAlign)
20621 return Header->getAlignment(); // Already processed.
20622
20623 unsigned LoopSize = 0;
20624 for (const MachineBasicBlock *MBB : ML->blocks()) {
20625 // If inner loop block is aligned assume in average half of the alignment
20626 // size to be added as nops.
20627 if (MBB != Header)
20628 LoopSize += MBB->getAlignment().value() / 2;
20629
20630 for (const MachineInstr &MI : *MBB) {
20631 LoopSize += TII->getInstSizeInBytes(MI);
20632 if (LoopSize > 192)
20633 return PrefAlign;
20634 }
20635 }
20636
20637 if (LoopSize <= 64)
20638 return PrefAlign;
20639
20640 if (LoopSize <= 128)
20641 return CacheLineAlign;
20642
20643 // If any of parent loops is surrounded by prefetch instructions do not
20644 // insert new for inner loop, which would reset parent's settings.
20645 for (MachineLoop *P = ML->getParentLoop(); P; P = P->getParentLoop()) {
20646 if (MachineBasicBlock *Exit = P->getExitBlock()) {
20647 auto I = Exit->getFirstNonDebugInstr();
20648 if (I != Exit->end() && I->getOpcode() == AMDGPU::S_INST_PREFETCH)
20649 return CacheLineAlign;
20650 }
20651 }
20652
20653 MachineBasicBlock *Pre = ML->getLoopPreheader();
20654 MachineBasicBlock *Exit = ML->getExitBlock();
20655
20656 if (Pre && Exit) {
20657 auto PreTerm = Pre->getFirstTerminator();
20658 if (PreTerm == Pre->begin() ||
20659 std::prev(PreTerm)->getOpcode() != AMDGPU::S_INST_PREFETCH)
20660 BuildMI(*Pre, PreTerm, DebugLoc(), TII->get(AMDGPU::S_INST_PREFETCH))
20661 .addImm(1); // prefetch 2 lines behind PC
20662
20663 auto ExitHead = Exit->getFirstNonDebugInstr();
20664 if (ExitHead == Exit->end() ||
20665 ExitHead->getOpcode() != AMDGPU::S_INST_PREFETCH)
20666 BuildMI(*Exit, ExitHead, DebugLoc(), TII->get(AMDGPU::S_INST_PREFETCH))
20667 .addImm(2); // prefetch 1 line behind PC
20668 }
20669
20670 return CacheLineAlign;
20671}
20672
20674 MachineBasicBlock *MBB) const {
20675 // GFX950: Limit padding to 4 bytes (one s_nop) for blocks where an 8-byte
20676 // instruction could be split by the 32-byte fetch window boundary.
20677 // See getPrefLoopAlignment() for context.
20678 if (needsFetchWindowAlignment(*MBB))
20679 return 4;
20681}
20682
20683bool SITargetLowering::needsFetchWindowAlignment(
20684 const MachineBasicBlock &MBB) const {
20685 if (!getSubtarget()->hasLoopHeadInstSplitSensitivity())
20686 return false;
20688 for (const MachineInstr &MI : MBB) {
20689 if (MI.isMetaInstruction())
20690 continue;
20691 // Instructions larger than 4 bytes can be split by a 32-byte boundary.
20692 return TII->getInstSizeInBytes(MI) > 4;
20693 }
20694 return false;
20695}
20696
20697[[maybe_unused]]
20698static bool isCopyFromRegOfInlineAsm(const SDNode *N) {
20699 assert(N->getOpcode() == ISD::CopyFromReg);
20700 do {
20701 // Follow the chain until we find an INLINEASM node.
20702 N = N->getOperand(0).getNode();
20703 if (N->getOpcode() == ISD::INLINEASM || N->getOpcode() == ISD::INLINEASM_BR)
20704 return true;
20705 } while (N->getOpcode() == ISD::CopyFromReg);
20706 return false;
20707}
20708
20711 UniformityInfo *UA) const {
20712 switch (N->getOpcode()) {
20713 case ISD::CopyFromReg: {
20714 const RegisterSDNode *R = cast<RegisterSDNode>(N->getOperand(1));
20715 const MachineRegisterInfo &MRI = FLI->MF->getRegInfo();
20716 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
20717 Register Reg = R->getReg();
20718
20719 // FIXME: Why does this need to consider isLiveIn?
20720 if (Reg.isPhysical() || MRI.isLiveIn(Reg))
20721 return !TRI->isSGPRReg(MRI, Reg);
20722
20723 if (const Value *V = FLI->getValueFromVirtualReg(R->getReg()))
20724 return UA->isDivergentAtDef(V);
20725
20727 return !TRI->isSGPRReg(MRI, Reg);
20728 }
20729 case ISD::LOAD: {
20730 const LoadSDNode *L = cast<LoadSDNode>(N);
20731 unsigned AS = L->getAddressSpace();
20732 // A flat load may access private memory.
20734 }
20735 case ISD::CALLSEQ_END:
20736 return true;
20738 return AMDGPU::isIntrinsicSourceOfDivergence(N->getConstantOperandVal(0));
20740 return AMDGPU::isIntrinsicSourceOfDivergence(N->getConstantOperandVal(1));
20741 case AMDGPUISD::ATOMIC_CMP_SWAP:
20742 case AMDGPUISD::BUFFER_ATOMIC_SWAP:
20743 case AMDGPUISD::BUFFER_ATOMIC_ADD:
20744 case AMDGPUISD::BUFFER_ATOMIC_SUB:
20745 case AMDGPUISD::BUFFER_ATOMIC_SMIN:
20746 case AMDGPUISD::BUFFER_ATOMIC_UMIN:
20747 case AMDGPUISD::BUFFER_ATOMIC_SMAX:
20748 case AMDGPUISD::BUFFER_ATOMIC_UMAX:
20749 case AMDGPUISD::BUFFER_ATOMIC_AND:
20750 case AMDGPUISD::BUFFER_ATOMIC_OR:
20751 case AMDGPUISD::BUFFER_ATOMIC_XOR:
20752 case AMDGPUISD::BUFFER_ATOMIC_INC:
20753 case AMDGPUISD::BUFFER_ATOMIC_DEC:
20754 case AMDGPUISD::BUFFER_ATOMIC_CMPSWAP:
20755 case AMDGPUISD::BUFFER_ATOMIC_FADD:
20756 case AMDGPUISD::BUFFER_ATOMIC_FMIN:
20757 case AMDGPUISD::BUFFER_ATOMIC_FMAX:
20758 // Target-specific read-modify-write atomics are sources of divergence.
20759 return true;
20760 default:
20761 if (auto *A = dyn_cast<AtomicSDNode>(N)) {
20762 // Generic read-modify-write atomics are sources of divergence.
20763 return A->readMem() && A->writeMem();
20764 }
20765 return false;
20766 }
20767}
20768
20770 EVT VT) const {
20771 switch (VT.getScalarType().getSimpleVT().SimpleTy) {
20772 case MVT::f32:
20774 case MVT::f64:
20775 case MVT::f16:
20777 default:
20778 return false;
20779 }
20780}
20781
20783 LLT Ty, const MachineFunction &MF) const {
20784 switch (Ty.getScalarSizeInBits()) {
20785 case 32:
20786 return !denormalModeIsFlushAllF32(MF);
20787 case 64:
20788 case 16:
20789 return !denormalModeIsFlushAllF64F16(MF);
20790 default:
20791 return false;
20792 }
20793}
20794
20796 const APInt &DemandedElts,
20797 const SelectionDAG &DAG,
20798 bool SNaN,
20799 unsigned Depth) const {
20800 if (Op.getOpcode() == AMDGPUISD::CLAMP) {
20801 const MachineFunction &MF = DAG.getMachineFunction();
20803
20804 if (Info->getMode().DX10Clamp)
20805 return true; // Clamped to 0.
20806 return DAG.isKnownNeverNaN(Op.getOperand(0), SNaN, Depth + 1);
20807 }
20808
20810 DAG, SNaN, Depth);
20811}
20812
20813namespace {
20814
20815/// Why a floating-point atomic instruction which may flush denormals is
20816/// acceptable.
20817enum class AtomicFlushDenormalReason {
20818 Native,
20819 IEEE,
20820 IgnoreDenormalMode,
20821 FunctionFlushesDenormals
20822};
20823
20824/// Why a native floating-point atomic instruction is acceptable for a global
20825/// memory address.
20826enum class GlobalFPAtomicLegality {
20827 Illegal,
20828 AgentScopeFineGrainedRemoteMemory,
20829 EmulatedSystemScope,
20830 NoRemoteMemory,
20831 NoFineGrainedMemory
20832};
20833
20834} // end anonymous namespace
20835
20836// On older subtargets, global FP atomic instructions have a hardcoded FP mode
20837// and do not support FP32 denormals, and only support v2f16/f64 denormals.
20838static AtomicFlushDenormalReason
20840 if (RMW->hasMetadata(LLVMContext::MD_atomic_ignore_denormal_mode))
20841 return AtomicFlushDenormalReason::IgnoreDenormalMode;
20842
20843 const fltSemantics &Flt = RMW->getType()->getScalarType()->getFltSemantics();
20844 auto DenormMode = RMW->getFunction()->getDenormalMode(Flt);
20845 return DenormMode == DenormalMode::getPreserveSign()
20846 ? AtomicFlushDenormalReason::FunctionFlushesDenormals
20847 : AtomicFlushDenormalReason::IEEE;
20848}
20849
20850static OptimizationRemark
20852 GlobalFPAtomicLegality MemLegality,
20853 AtomicFlushDenormalReason DenormReason) {
20854 LLVMContext &Ctx = RMW->getContext();
20855 StringRef MemScope = Ctx.getSyncScopeName(RMW->getSyncScopeID()).value_or("");
20856 if (MemScope.empty())
20857 MemScope = "system";
20858
20859 OptimizationRemark R(DEBUG_TYPE, "Passed", RMW);
20860 R << "hardware instruction generated for atomic "
20861 << ore::NV("Operation", RMW->getOperationName(RMW->getOperation()))
20862 << " at " << ore::NV("SyncScope", MemScope) << " scope since ";
20863
20864 switch (MemLegality) {
20865 case GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory:
20866 R << "fine-grained remote memory atomics work below system scope";
20867 break;
20868 case GlobalFPAtomicLegality::EmulatedSystemScope:
20869 R << "system scope atomics are emulated in hardware";
20870 break;
20871 case GlobalFPAtomicLegality::NoRemoteMemory:
20872 R << "memory is not remote (!amdgpu.no.remote.memory)";
20873 break;
20874 case GlobalFPAtomicLegality::NoFineGrainedMemory:
20875 R << "memory is not fine-grained (!amdgpu.no.fine.grained.memory)";
20876 break;
20877 case GlobalFPAtomicLegality::Illegal:
20878 llvm_unreachable("remark for illegal atomic");
20879 }
20880
20881 switch (DenormReason) {
20882 case AtomicFlushDenormalReason::Native:
20883 break;
20884 case AtomicFlushDenormalReason::IgnoreDenormalMode:
20885 R << ", and denormals may be flushed (!atomic.ignore.denormal.mode)";
20886 break;
20887 case AtomicFlushDenormalReason::FunctionFlushesDenormals:
20888 R << ", and the floating-point environment flushes denormals";
20889 break;
20890 case AtomicFlushDenormalReason::IEEE:
20891 llvm_unreachable("remark for illegal atomic");
20892 }
20893
20894 return R;
20895}
20896
20897static bool isV2F16OrV2BF16(Type *Ty) {
20898 if (auto *VT = dyn_cast<FixedVectorType>(Ty)) {
20899 Type *EltTy = VT->getElementType();
20900 return VT->getNumElements() == 2 &&
20901 (EltTy->isHalfTy() || EltTy->isBFloatTy());
20902 }
20903
20904 return false;
20905}
20906
20907static bool isV2F16(Type *Ty) {
20909 return VT && VT->getNumElements() == 2 && VT->getElementType()->isHalfTy();
20910}
20911
20912static bool isV2BF16(Type *Ty) {
20914 return VT && VT->getNumElements() == 2 && VT->getElementType()->isBFloatTy();
20915}
20916
20917/// \return true if atomicrmw integer ops work for the type.
20918static bool isAtomicRMWLegalIntTy(Type *Ty) {
20919 if (auto *IT = dyn_cast<IntegerType>(Ty)) {
20920 unsigned BW = IT->getBitWidth();
20921 return BW == 32 || BW == 64;
20922 }
20923
20924 return false;
20925}
20926
20927/// \return true if this atomicrmw xchg type can be selected.
20928static bool isAtomicRMWLegalXChgTy(const AtomicRMWInst *RMW) {
20929 Type *Ty = RMW->getType();
20930 if (isAtomicRMWLegalIntTy(Ty))
20931 return true;
20932
20933 if (PointerType *PT = dyn_cast<PointerType>(Ty)) {
20934 const DataLayout &DL = RMW->getFunction()->getDataLayout();
20935 unsigned BW = DL.getPointerSizeInBits(PT->getAddressSpace());
20936 return BW == 32 || BW == 64;
20937 }
20938
20939 if (Ty->isFloatTy() || Ty->isDoubleTy())
20940 return true;
20941
20943 return VT->getNumElements() == 2 &&
20944 VT->getElementType()->getPrimitiveSizeInBits() == 16;
20945 }
20946
20947 return false;
20948}
20949
20950/// \returns whether it's valid to emit a native instruction for \p RMW, and
20951/// why, based on the properties of the target memory.
20952static GlobalFPAtomicLegality
20954 const AtomicRMWInst *RMW, bool HasSystemScope) {
20955 // The remote/fine-grained access logic is different from the integer
20956 // atomics. Without AgentScopeFineGrainedRemoteMemoryAtomics support,
20957 // fine-grained access does not work, even for a device local allocation.
20958 //
20959 // With AgentScopeFineGrainedRemoteMemoryAtomics, system scoped device local
20960 // allocations work.
20961 if (HasSystemScope) {
20962 if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics() &&
20963 RMW->hasMetadata("amdgpu.no.remote.memory"))
20964 return GlobalFPAtomicLegality::NoRemoteMemory;
20965 if (Subtarget.hasEmulatedSystemScopeAtomics())
20966 return GlobalFPAtomicLegality::EmulatedSystemScope;
20967 } else if (Subtarget.hasAgentScopeFineGrainedRemoteMemoryAtomics())
20968 return GlobalFPAtomicLegality::AgentScopeFineGrainedRemoteMemory;
20969
20970 return RMW->hasMetadata("amdgpu.no.fine.grained.memory")
20971 ? GlobalFPAtomicLegality::NoFineGrainedMemory
20972 : GlobalFPAtomicLegality::Illegal;
20973}
20974
20975/// \return Action to perform on AtomicRMWInsts for integer operations.
20982
20983/// Return if a flat address space atomicrmw can access private memory.
20985 const MDNode *MD = I->getMetadata(LLVMContext::MD_noalias_addrspace);
20986 return !MD ||
20988}
20989
20992 // For GAS, lower to flat atomic.
20993 return STI.hasGloballyAddressableScratch()
20996}
20997
21000 unsigned AS = RMW->getPointerAddressSpace();
21001 if (AS == AMDGPUAS::PRIVATE_ADDRESS)
21003
21004 // 64-bit flat atomics that dynamically reside in private memory will silently
21005 // be dropped.
21006 //
21007 // Note that we will emit a new copy of the original atomic in the expansion,
21008 // which will be incrementally relegalized.
21009 const DataLayout &DL = RMW->getFunction()->getDataLayout();
21010 if (AS == AMDGPUAS::FLAT_ADDRESS &&
21011 DL.getTypeSizeInBits(RMW->getType()) == 64 &&
21014
21015 GlobalFPAtomicLegality MemLegality = GlobalFPAtomicLegality::Illegal;
21016 AtomicFlushDenormalReason DenormReason = AtomicFlushDenormalReason::Native;
21017 auto ReportHWInst = [&](TargetLowering::AtomicExpansionKind Kind) {
21019 ORE.emit([&]() {
21020 return emitAtomicRMWLegalRemark(RMW, MemLegality, DenormReason);
21021 });
21022 return Kind;
21023 };
21024
21025 auto SSID = RMW->getSyncScopeID();
21026 bool HasSystemScope =
21027 SSID == SyncScope::System ||
21029 getTargetMachine().getTargetTriple(), AtomicScope::System,
21030 /*OneAddressSpace=*/true));
21031
21032 auto Op = RMW->getOperation();
21033 switch (Op) {
21035 // PCIe supports add and xchg for system atomics.
21036 return isAtomicRMWLegalXChgTy(RMW)
21039 case AtomicRMWInst::Add:
21040 // PCIe supports add and xchg for system atomics.
21042 case AtomicRMWInst::Sub:
21043 case AtomicRMWInst::And:
21044 case AtomicRMWInst::Or:
21045 case AtomicRMWInst::Xor:
21046 case AtomicRMWInst::Max:
21047 case AtomicRMWInst::Min:
21054 if (Op == AtomicRMWInst::USubCond && !Subtarget->hasCondSubInsts())
21056 if (Op == AtomicRMWInst::USubSat) {
21057 // The global and buffer forms predate the LDS and flat ones.
21058 if (!Subtarget->hasSubClampInsts() ||
21059 (AS == AMDGPUAS::LOCAL_ADDRESS &&
21060 !Subtarget->hasAtomicDsCondSubClampInsts()) ||
21061 (AS == AMDGPUAS::FLAT_ADDRESS &&
21062 !Subtarget->hasAtomicCondSubClampFlatInsts()))
21064 }
21066 auto *IT = dyn_cast<IntegerType>(RMW->getType());
21067 if (!IT || IT->getBitWidth() != 32)
21069 }
21070
21073 if (Subtarget->hasEmulatedSystemScopeAtomics())
21075
21076 // On most subtargets, for atomicrmw operations other than add/xchg,
21077 // whether or not the instructions will behave correctly depends on where
21078 // the address physically resides and what interconnect is used in the
21079 // system configuration. On some some targets the instruction will nop,
21080 // and in others synchronization will only occur at degraded device scope.
21081 //
21082 // If the allocation is known local to the device, the instructions should
21083 // work correctly.
21084 if (RMW->hasMetadata("amdgpu.no.remote.memory"))
21086
21087 // If fine-grained remote memory works at device scope, we don't need to
21088 // do anything.
21089 if (!HasSystemScope &&
21090 Subtarget->hasAgentScopeFineGrainedRemoteMemoryAtomics())
21092
21093 // If we are targeting a remote allocated address, it depends what kind of
21094 // allocation the address belongs to.
21095 //
21096 // If the allocation is fine-grained (in host memory, or in PCIe peer
21097 // device memory), the operation will fail depending on the target.
21098 //
21099 // Note fine-grained host memory access does work on APUs or if XGMI is
21100 // used, but we do not know if we are targeting an APU or the system
21101 // configuration from the ISA version/target-cpu.
21102 if (RMW->hasMetadata("amdgpu.no.fine.grained.memory"))
21104
21107 // Atomic sub/or/xor do not work over PCI express, but atomic add
21108 // does. InstCombine transforms these with 0 to or, so undo that.
21109 // Sub-word types are not selectable and take the cmpxchg expansion.
21110 if (const Constant *ConstVal = dyn_cast<Constant>(RMW->getValOperand());
21111 ConstVal && ConstVal->isNullValue() &&
21114 }
21115
21116 // If the allocation could be in remote, fine-grained memory, the rmw
21117 // instructions may fail. cmpxchg should work, so emit that. On some
21118 // system configurations, PCIe atomics aren't supported so cmpxchg won't
21119 // even work, so you're out of luck anyway.
21120
21121 // In summary:
21122 //
21123 // Cases that may fail:
21124 // - fine-grained pinned host memory
21125 // - fine-grained migratable host memory
21126 // - fine-grained PCIe peer device
21127 //
21128 // Cases that should work, but may be treated overly conservatively.
21129 // - fine-grained host memory on an APU
21130 // - fine-grained XGMI peer device
21132 }
21133
21135 }
21136 case AtomicRMWInst::FAdd: {
21137 Type *Ty = RMW->getType();
21138
21139 // TODO: Handle REGION_ADDRESS
21140 if (AS == AMDGPUAS::LOCAL_ADDRESS) {
21141 // DS F32 FP atomics do respect the denormal mode, but the rounding mode
21142 // is fixed to round-to-nearest-even.
21143 //
21144 // F64 / PK_F16 / PK_BF16 never flush and are also fixed to
21145 // round-to-nearest-even.
21146 //
21147 // We ignore the rounding mode problem, even in strictfp. The C++ standard
21148 // suggests it is OK if the floating-point mode may not match the calling
21149 // thread.
21150 if (Ty->isFloatTy()) {
21151 return Subtarget->hasLDSFPAtomicAddF32() ? AtomicExpansionKind::None
21153 }
21154
21155 if (Ty->isDoubleTy()) {
21156 // Ignores denormal mode, but we don't consider flushing mandatory.
21157 return Subtarget->hasLDSFPAtomicAddF64() ? AtomicExpansionKind::None
21159 }
21160
21161 if (Subtarget->hasAtomicDsPkAdd16Insts() && isV2F16OrV2BF16(Ty))
21163
21165 }
21166
21167 // LDS atomics respect the denormal mode from the mode register.
21168 //
21169 // Traditionally f32 global/buffer memory atomics would unconditionally
21170 // flush denormals, but newer targets do not flush. f64/f16/bf16 cases never
21171 // flush.
21172 //
21173 // On targets with flat atomic fadd, denormals would flush depending on
21174 // whether the target address resides in LDS or global memory. We consider
21175 // this flat-maybe-flush as will-flush.
21176 if (Ty->isFloatTy() &&
21177 !Subtarget->hasMemoryAtomicFaddF32DenormalSupport()) {
21178 DenormReason = getAtomicFlushDenormalReason(RMW);
21179 if (DenormReason == AtomicFlushDenormalReason::IEEE)
21181 }
21182
21183 MemLegality =
21184 getGlobalMemoryFPAtomicLegality(*Subtarget, RMW, HasSystemScope);
21185 if (MemLegality != GlobalFPAtomicLegality::Illegal) {
21186 if (AS == AMDGPUAS::FLAT_ADDRESS) {
21187 // gfx942, gfx12
21188 if (Subtarget->hasAtomicFlatPkAdd16Insts() && isV2F16OrV2BF16(Ty))
21189 return ReportHWInst(AtomicExpansionKind::None);
21190 } else if (AMDGPU::isExtendedGlobalAddrSpace(AS)) {
21191 // gfx90a, gfx942, gfx12
21192 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() && isV2F16(Ty))
21193 return ReportHWInst(AtomicExpansionKind::None);
21194
21195 // gfx942, gfx12
21196 if (Subtarget->hasAtomicGlobalPkAddBF16Inst() && isV2BF16(Ty))
21197 return ReportHWInst(AtomicExpansionKind::None);
21198 } else if (AS == AMDGPUAS::BUFFER_FAT_POINTER) {
21199 // gfx90a, gfx942, gfx12
21200 if (Subtarget->hasAtomicBufferGlobalPkAddF16Insts() && isV2F16(Ty))
21201 return ReportHWInst(AtomicExpansionKind::None);
21202
21203 // While gfx90a/gfx942 supports v2bf16 for global/flat, it does not for
21204 // buffer. gfx12 does have the buffer version.
21205 if (Subtarget->hasAtomicBufferPkAddBF16Inst() && isV2BF16(Ty))
21206 return ReportHWInst(AtomicExpansionKind::None);
21207 }
21208
21209 // global and flat atomic fadd f64: gfx90a, gfx942.
21210 if (Subtarget->hasFlatBufferGlobalAtomicFaddF64Inst() && Ty->isDoubleTy())
21211 return ReportHWInst(AtomicExpansionKind::None);
21212
21213 if (AS != AMDGPUAS::FLAT_ADDRESS) {
21214 if (Ty->isFloatTy()) {
21215 // global/buffer atomic fadd f32 no-rtn: gfx908, gfx90a, gfx942,
21216 // gfx11+.
21217 if (RMW->use_empty() && Subtarget->hasAtomicFaddNoRtnInsts())
21218 return ReportHWInst(AtomicExpansionKind::None);
21219 // global/buffer atomic fadd f32 rtn: gfx90a, gfx942, gfx11+.
21220 if (!RMW->use_empty() && Subtarget->hasAtomicFaddRtnInsts())
21221 return ReportHWInst(AtomicExpansionKind::None);
21222 } else {
21223 // gfx908
21224 if (RMW->use_empty() &&
21225 Subtarget->hasAtomicBufferGlobalPkAddF16NoRtnInsts() &&
21226 isV2F16(Ty))
21227 return ReportHWInst(AtomicExpansionKind::None);
21228 }
21229 }
21230
21231 // flat atomic fadd f32: gfx942, gfx11+.
21232 if (AS == AMDGPUAS::FLAT_ADDRESS && Ty->isFloatTy()) {
21233 if (Subtarget->hasFlatAtomicFaddF32Inst())
21234 return ReportHWInst(AtomicExpansionKind::None);
21235
21236 // If it is in flat address space, and the type is float, we will try to
21237 // expand it, if the target supports global and lds atomic fadd. The
21238 // reason we need that is, in the expansion, we emit the check of
21239 // address space. If it is in global address space, we emit the global
21240 // atomic fadd; if it is in shared address space, we emit the LDS atomic
21241 // fadd.
21242 if (Subtarget->hasLDSFPAtomicAddF32()) {
21243 if (RMW->use_empty() && Subtarget->hasAtomicFaddNoRtnInsts())
21245 if (!RMW->use_empty() && Subtarget->hasAtomicFaddRtnInsts())
21247 }
21248 }
21249 }
21250
21252 }
21254 case AtomicRMWInst::FMax: {
21255 Type *Ty = RMW->getType();
21256
21257 // LDS float and double fmin/fmax were always supported.
21258 if (AS == AMDGPUAS::LOCAL_ADDRESS) {
21259 return Ty->isFloatTy() || Ty->isDoubleTy() ? AtomicExpansionKind::None
21261 }
21262
21263 MemLegality =
21264 getGlobalMemoryFPAtomicLegality(*Subtarget, RMW, HasSystemScope);
21265 if (MemLegality != GlobalFPAtomicLegality::Illegal) {
21266 // For flat and global cases:
21267 // float, double in gfx7. Manual claims denormal support.
21268 // Removed in gfx8.
21269 // float, double restored in gfx10.
21270 // double removed again in gfx11, so only f32 for gfx11/gfx12.
21271 //
21272 // For gfx9, gfx90a and gfx942 support f64 for global (same as fadd), but
21273 // no f32.
21274 if (AS == AMDGPUAS::FLAT_ADDRESS) {
21275 if (Subtarget->hasAtomicFMinFMaxF32FlatInsts() && Ty->isFloatTy())
21276 return ReportHWInst(AtomicExpansionKind::None);
21277 if (Subtarget->hasAtomicFMinFMaxF64FlatInsts() && Ty->isDoubleTy())
21278 return ReportHWInst(AtomicExpansionKind::None);
21279 } else if (AMDGPU::isExtendedGlobalAddrSpace(AS) ||
21281 if (Subtarget->hasAtomicFMinFMaxF32GlobalInsts() && Ty->isFloatTy())
21282 return ReportHWInst(AtomicExpansionKind::None);
21283 if (Subtarget->hasAtomicFMinFMaxF64GlobalInsts() && Ty->isDoubleTy())
21284 return ReportHWInst(AtomicExpansionKind::None);
21285 }
21286 }
21287
21289 }
21292 default:
21294 }
21295
21296 llvm_unreachable("covered atomicrmw op switch");
21297}
21298
21305
21312
21315 const AtomicCmpXchgInst *CmpX) const {
21316 unsigned AddrSpace = CmpX->getPointerAddressSpace();
21317 if (AddrSpace == AMDGPUAS::PRIVATE_ADDRESS)
21319
21320 if (AddrSpace != AMDGPUAS::FLAT_ADDRESS || !flatInstrMayAccessPrivate(CmpX))
21322
21323 const DataLayout &DL = CmpX->getDataLayout();
21324
21325 Type *ValTy = CmpX->getNewValOperand()->getType();
21326
21327 // If a 64-bit flat atomic may alias private, we need to avoid using the
21328 // atomic in the private case.
21329 return DL.getTypeSizeInBits(ValTy) == 64 ? AtomicExpansionKind::CustomExpand
21331}
21332
21333const TargetRegisterClass *
21334SITargetLowering::getRegClassFor(MVT VT, bool isDivergent) const {
21336 const SIRegisterInfo *TRI = Subtarget->getRegisterInfo();
21337 if (RC == &AMDGPU::VReg_1RegClass && !isDivergent)
21338 return Subtarget->isWave64() ? &AMDGPU::SReg_64RegClass
21339 : &AMDGPU::SReg_32RegClass;
21340 if (!TRI->isSGPRClass(RC) && !isDivergent)
21341 return TRI->getEquivalentSGPRClass(RC);
21342 if (TRI->isSGPRClass(RC) && isDivergent) {
21343 if (Subtarget->hasGFX90AInsts())
21344 return TRI->getEquivalentAVClass(RC);
21345 return TRI->getEquivalentVGPRClass(RC);
21346 }
21347
21348 return RC;
21349}
21350
21351// FIXME: This is a workaround for DivergenceAnalysis not understanding always
21352// uniform values (as produced by the mask results of control flow intrinsics)
21353// used outside of divergent blocks. The phi users need to also be treated as
21354// always uniform.
21355//
21356// FIXME: DA is no longer in-use. Does this still apply to UniformityAnalysis?
21357static bool hasCFUser(const Value *V, SmallPtrSet<const Value *, 16> &Visited,
21358 unsigned WaveSize) {
21359 // FIXME: We assume we never cast the mask results of a control flow
21360 // intrinsic.
21361 // Early exit if the type won't be consistent as a compile time hack.
21362 IntegerType *IT = dyn_cast<IntegerType>(V->getType());
21363 if (!IT || IT->getBitWidth() != WaveSize)
21364 return false;
21365
21366 if (!isa<Instruction>(V))
21367 return false;
21368 if (!Visited.insert(V).second)
21369 return false;
21370 bool Result = false;
21371 for (const auto *U : V->users()) {
21373 if (V == U->getOperand(1)) {
21374 switch (Intrinsic->getIntrinsicID()) {
21375 default:
21376 Result = false;
21377 break;
21378 case Intrinsic::amdgcn_if_break:
21379 case Intrinsic::amdgcn_if:
21380 case Intrinsic::amdgcn_else:
21381 Result = true;
21382 break;
21383 }
21384 }
21385 if (V == U->getOperand(0)) {
21386 switch (Intrinsic->getIntrinsicID()) {
21387 default:
21388 Result = false;
21389 break;
21390 case Intrinsic::amdgcn_end_cf:
21391 case Intrinsic::amdgcn_loop:
21392 Result = true;
21393 break;
21394 }
21395 }
21396 } else {
21397 Result = hasCFUser(U, Visited, WaveSize);
21398 }
21399 if (Result)
21400 break;
21401 }
21402 return Result;
21403}
21404
21406 const Value *V) const {
21407 if (const CallInst *CI = dyn_cast<CallInst>(V)) {
21408 if (CI->isInlineAsm()) {
21409 // FIXME: This cannot give a correct answer. This should only trigger in
21410 // the case where inline asm returns mixed SGPR and VGPR results, used
21411 // outside the defining block. We don't have a specific result to
21412 // consider, so this assumes if any value is SGPR, the overall register
21413 // also needs to be SGPR.
21414 const SIRegisterInfo *SIRI = Subtarget->getRegisterInfo();
21416 MF.getDataLayout(), Subtarget->getRegisterInfo(), *CI);
21417 for (auto &TC : TargetConstraints) {
21418 if (TC.Type == InlineAsm::isOutput) {
21420 const TargetRegisterClass *RC =
21421 getRegForInlineAsmConstraint(SIRI, TC.ConstraintCode,
21422 TC.ConstraintVT)
21423 .second;
21424 if (RC && SIRI->isSGPRClass(RC))
21425 return true;
21426 }
21427 }
21428 }
21429 }
21431 return hasCFUser(V, Visited, Subtarget->getWavefrontSize());
21432}
21433
21435 for (SDUse &Use : N->uses()) {
21437 if (getBasePtrIndex(M) == Use.getOperandNo())
21438 return true;
21439 }
21440 }
21441 return false;
21442}
21443
21445 SDValue N1) const {
21446 if (!N0.hasOneUse())
21447 return false;
21448 // Take care of the opportunity to keep N0 uniform
21449 if (N0->isDivergent() || !N1->isDivergent())
21450 return true;
21451 // Check if we have a good chance to form the memory access pattern with the
21452 // base and offset
21453 return (DAG.isBaseWithConstantOffset(N0) &&
21455}
21456
21458 Register N0, Register N1) const {
21459 return MRI.hasOneNonDBGUse(N0); // FIXME: handle regbanks
21460}
21461
21464 // Propagate metadata set by AMDGPUAnnotateUniformValues to the MMO of a load.
21466 if (I.getMetadata("amdgpu.noclobber"))
21467 Flags |= MONoClobber;
21468 if (I.getMetadata("amdgpu.last.use"))
21469 Flags |= MOLastUse;
21470 return Flags;
21471}
21472
21474 Instruction *AI) const {
21475 // Given: atomicrmw fadd ptr %addr, float %val ordering
21476 //
21477 // With this expansion we produce the following code:
21478 // [...]
21479 // %is.shared = call i1 @llvm.amdgcn.is.shared(ptr %addr)
21480 // br i1 %is.shared, label %atomicrmw.shared, label %atomicrmw.check.private
21481 //
21482 // atomicrmw.shared:
21483 // %cast.shared = addrspacecast ptr %addr to ptr addrspace(3)
21484 // %loaded.shared = atomicrmw fadd ptr addrspace(3) %cast.shared,
21485 // float %val ordering
21486 // br label %atomicrmw.phi
21487 //
21488 // atomicrmw.check.private:
21489 // %is.private = call i1 @llvm.amdgcn.is.private(ptr %int8ptr)
21490 // br i1 %is.private, label %atomicrmw.private, label %atomicrmw.global
21491 //
21492 // atomicrmw.private:
21493 // %cast.private = addrspacecast ptr %addr to ptr addrspace(5)
21494 // %loaded.private = load float, ptr addrspace(5) %cast.private
21495 // %val.new = fadd float %loaded.private, %val
21496 // store float %val.new, ptr addrspace(5) %cast.private
21497 // br label %atomicrmw.phi
21498 //
21499 // atomicrmw.global:
21500 // %cast.global = addrspacecast ptr %addr to ptr addrspace(1)
21501 // %loaded.global = atomicrmw fadd ptr addrspace(1) %cast.global,
21502 // float %val ordering
21503 // br label %atomicrmw.phi
21504 //
21505 // atomicrmw.phi:
21506 // %loaded.phi = phi float [ %loaded.shared, %atomicrmw.shared ],
21507 // [ %loaded.private, %atomicrmw.private ],
21508 // [ %loaded.global, %atomicrmw.global ]
21509 // br label %atomicrmw.end
21510 //
21511 // atomicrmw.end:
21512 // [...]
21513 //
21514 //
21515 // For 64-bit atomics which may reside in private memory, we perform a simpler
21516 // version that only inserts the private check, and uses the flat operation.
21517
21518 IRBuilder<> Builder(AI);
21519 LLVMContext &Ctx = Builder.getContext();
21520
21521 auto *RMW = dyn_cast<AtomicRMWInst>(AI);
21522 const unsigned PtrOpIdx = RMW ? AtomicRMWInst::getPointerOperandIndex()
21524 Value *Addr = AI->getOperand(PtrOpIdx);
21525
21526 /// TODO: Only need to check private, then emit flat-known-not private (no
21527 /// need for shared block, or cast to global).
21529
21530 Align Alignment;
21531 if (RMW)
21532 Alignment = RMW->getAlign();
21533 else if (CX)
21534 Alignment = CX->getAlign();
21535 else
21536 llvm_unreachable("unhandled atomic operation");
21537
21538 // FullFlatEmulation is true if we need to issue the private, shared, and
21539 // global cases.
21540 //
21541 // If this is false, we are only dealing with the flat-targeting-private case,
21542 // where we only insert a check for private and still use the flat instruction
21543 // for global and shared.
21544
21545 bool FullFlatEmulation =
21546 RMW && RMW->getOperation() == AtomicRMWInst::FAdd &&
21547 ((Subtarget->hasAtomicFaddInsts() && RMW->getType()->isFloatTy()) ||
21548 (Subtarget->hasFlatBufferGlobalAtomicFaddF64Inst() &&
21549 RMW->getType()->isDoubleTy()));
21550
21551 // If the return value isn't used, do not introduce a false use in the phi.
21552 bool ReturnValueIsUsed = !AI->use_empty();
21553
21554 BasicBlock *BB = Builder.GetInsertBlock();
21555 Function *F = BB->getParent();
21556 BasicBlock *ExitBB =
21557 BB->splitBasicBlock(Builder.GetInsertPoint(), "atomicrmw.end");
21558 BasicBlock *SharedBB = nullptr;
21559
21560 BasicBlock *CheckPrivateBB = BB;
21561 if (FullFlatEmulation) {
21562 SharedBB = BasicBlock::Create(Ctx, "atomicrmw.shared", F, ExitBB);
21563 CheckPrivateBB =
21564 BasicBlock::Create(Ctx, "atomicrmw.check.private", F, ExitBB);
21565 }
21566
21567 BasicBlock *PrivateBB =
21568 BasicBlock::Create(Ctx, "atomicrmw.private", F, ExitBB);
21569 BasicBlock *GlobalBB = BasicBlock::Create(Ctx, "atomicrmw.global", F, ExitBB);
21570 BasicBlock *PhiBB = BasicBlock::Create(Ctx, "atomicrmw.phi", F, ExitBB);
21571
21572 std::prev(BB->end())->eraseFromParent();
21573 Builder.SetInsertPoint(BB);
21574
21575 Value *LoadedShared = nullptr;
21576 if (FullFlatEmulation) {
21577 Value *IsShared = Builder.CreateIntrinsic(Intrinsic::amdgcn_is_shared,
21578 {Addr}, nullptr, "is.shared");
21579 Builder.CreateCondBr(IsShared, SharedBB, CheckPrivateBB);
21580 Builder.SetInsertPoint(SharedBB);
21581 Value *CastToLocal = Builder.CreateAddrSpaceCast(
21583
21584 Instruction *Clone = AI->clone();
21585 Clone->insertInto(SharedBB, SharedBB->end());
21586 Clone->getOperandUse(PtrOpIdx).set(CastToLocal);
21587 LoadedShared = Clone;
21588
21589 Builder.CreateBr(PhiBB);
21590 Builder.SetInsertPoint(CheckPrivateBB);
21591 }
21592
21593 Value *IsPrivate = Builder.CreateIntrinsic(Intrinsic::amdgcn_is_private,
21594 {Addr}, nullptr, "is.private");
21595 Builder.CreateCondBr(IsPrivate, PrivateBB, GlobalBB);
21596
21597 Builder.SetInsertPoint(PrivateBB);
21598
21599 Value *CastToPrivate = Builder.CreateAddrSpaceCast(
21601
21602 Value *LoadedPrivate;
21603 if (RMW) {
21604 LoadedPrivate = Builder.CreateAlignedLoad(
21605 RMW->getType(), CastToPrivate, RMW->getAlign(), RMW->isVolatile(),
21606 "loaded.private");
21607
21608 Value *NewVal = buildAtomicRMWValue(RMW->getOperation(), Builder,
21609 LoadedPrivate, RMW->getValOperand());
21610
21611 Builder.CreateAlignedStore(NewVal, CastToPrivate, RMW->getAlign(),
21612 RMW->isVolatile());
21613 } else {
21614 auto [ResultLoad, Equal] = buildCmpXchgValue(
21615 Builder, CastToPrivate, CX->getCompareOperand(), CX->getNewValOperand(),
21616 CX->getAlign(), CX->isVolatile());
21617
21618 Value *Insert = Builder.CreateInsertValue(PoisonValue::get(CX->getType()),
21619 ResultLoad, 0);
21620 LoadedPrivate = Builder.CreateInsertValue(Insert, Equal, 1);
21621 }
21622
21623 Builder.CreateBr(PhiBB);
21624
21625 Builder.SetInsertPoint(GlobalBB);
21626
21627 // Continue using a flat instruction if we only emitted the check for private.
21628 Instruction *LoadedGlobal = AI;
21629 if (FullFlatEmulation) {
21630 Value *CastToGlobal = Builder.CreateAddrSpaceCast(
21632 AI->getOperandUse(PtrOpIdx).set(CastToGlobal);
21633 }
21634
21635 AI->removeFromParent();
21636 AI->insertInto(GlobalBB, GlobalBB->end());
21637
21638 // The new atomicrmw may go through another round of legalization later.
21639 if (!FullFlatEmulation) {
21640 // We inserted the runtime check already, make sure we do not try to
21641 // re-expand this.
21642 // TODO: Should union with any existing metadata.
21643 MDBuilder MDB(F->getContext());
21644 MDNode *RangeNotPrivate =
21647 LoadedGlobal->setMetadata(LLVMContext::MD_noalias_addrspace,
21648 RangeNotPrivate);
21649 }
21650
21651 Builder.CreateBr(PhiBB);
21652
21653 Builder.SetInsertPoint(PhiBB);
21654
21655 if (ReturnValueIsUsed) {
21656 PHINode *Loaded = Builder.CreatePHI(AI->getType(), 3);
21657 AI->replaceAllUsesWith(Loaded);
21658 if (FullFlatEmulation)
21659 Loaded->addIncoming(LoadedShared, SharedBB);
21660 Loaded->addIncoming(LoadedPrivate, PrivateBB);
21661 Loaded->addIncoming(LoadedGlobal, GlobalBB);
21662 Loaded->takeName(AI);
21663 }
21664
21665 Builder.CreateBr(ExitBB);
21666}
21667
21669 unsigned PtrOpIdx) {
21670 Value *PtrOp = I->getOperand(PtrOpIdx);
21673
21674 Type *FlatPtr = PointerType::get(I->getContext(), AMDGPUAS::FLAT_ADDRESS);
21675 Value *ASCast = CastInst::CreatePointerCast(PtrOp, FlatPtr, "scratch.ascast",
21676 I->getIterator());
21677 I->setOperand(PtrOpIdx, ASCast);
21678}
21679
21682
21685
21688 if (const auto *ConstVal = dyn_cast<Constant>(AI->getValOperand());
21689 ConstVal && ConstVal->isNullValue() &&
21691 // atomicrmw or %ptr, 0 -> atomicrmw add %ptr, 0
21693
21694 // We may still need the private-alias-flat handling below.
21695
21696 // TODO: Skip this for cases where we cannot access remote memory.
21697 }
21698 }
21699
21700 // The non-flat expansions should only perform the de-canonicalization of
21701 // identity values.
21703 return;
21704
21706}
21707
21714
21718
21720 "Expand Atomic Load only handles SCRATCH -> FLAT conversion");
21721}
21722
21724 if (SI->getPointerAddressSpace() == AMDGPUAS::PRIVATE_ADDRESS)
21725 return convertScratchAtomicToFlatAtomic(SI, SI->getPointerOperandIndex());
21726
21728 "Expand Atomic Store only handles SCRATCH -> FLAT conversion");
21729}
static bool isMul(MachineInstr *MI)
static unsigned getIntrinsicID(const SDNode *N)
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
AMDGPU address space definition.
static constexpr std::pair< ImplicitArgumentMask, StringLiteral > ImplicitAttrs[]
static bool allUsesHaveSourceMods(MachineInstr &MI, MachineRegisterInfo &MRI, unsigned CostThreshold=4)
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
static bool isCtlzOpc(unsigned Opc)
Contains the definition of a TargetInstrInfo class that is common to all AMD GPUs.
static bool isNoUnsignedWrap(MachineInstr *Addr)
static bool parseTexFail(uint64_t TexFailCtrl, bool &TFE, bool &LWE, bool &IsTexFail)
static bool isAsyncLDSDMA(Intrinsic::ID Intr)
static void packImage16bitOpsToDwords(MachineIRBuilder &B, MachineInstr &MI, SmallVectorImpl< Register > &PackedAddrs, unsigned ArgOffset, const AMDGPU::ImageDimIntrinsicInfo *Intr, bool IsA16, bool IsG16)
Turn a set of f16 typed registers in AddrRegs into a dword sized vector with f16 typed elements.
static bool isKnownNonNull(Register Val, MachineRegisterInfo &MRI, const AMDGPUTargetMachine &TM, unsigned AddrSpace)
Return true if the value is a known valid address, such that a null check is not necessary.
Provides AMDGPU specific target descriptions.
The AMDGPU TargetMachine interface definition for hw codegen targets.
This file declares a class to represent arbitrary precision floating point values and provide a varie...
This file implements a class to represent arbitrary precision integral constant values and operations...
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
MachineBasicBlock MachineBasicBlock::iterator MBBI
static cl::opt< ITMode > IT(cl::desc("IT block support"), cl::Hidden, cl::init(DefaultIT), cl::values(clEnumValN(DefaultIT, "arm-default-it", "Generate any type of IT block"), clEnumValN(RestrictedIT, "arm-restrict-it", "Disallow complex IT blocks")))
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
@ DEFAULT
Default weight is used in cases when there is no dedicated execution weight set.
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static std::optional< SDByteProvider > calculateByteProvider(SDValue Op, unsigned Index, unsigned Depth, std::optional< uint64_t > VectorIndex, unsigned StartingIndex=0, MutableArrayRef< uint8_t > ByteMask={})
dxil translate DXIL Translate Metadata
static bool isSigned(unsigned Opcode)
Utilities for dealing with flags related to floating point properties and mode controls.
AMD GCN specific subclass of TargetSubtarget.
Provides analysis for querying information about KnownBits during GISel passes.
#define DEBUG_TYPE
Declares convenience wrapper classes for interpreting MachineInstr instances as specific generic oper...
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
iv Induction Variable Users
Definition IVUsers.cpp:48
static constexpr Value * getValue(Ty &ValueOrUse)
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
Contains matchers for matching SSA Machine Instructions.
static bool isUndef(const MachineInstr &MI)
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
static unsigned getAddressSpace(const Value *V, unsigned MaxLookup)
uint64_t IntrinsicInst * II
#define P(N)
if(PassOpts->AAPipeline)
static constexpr MCPhysReg SPReg
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
Contains matchers for matching SelectionDAG nodes and values.
static void r0(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
Definition SHA1.cpp:39
static void r3(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
Definition SHA1.cpp:57
static void r2(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
Definition SHA1.cpp:51
static void r1(uint32_t &A, uint32_t &B, uint32_t &C, uint32_t &D, uint32_t &E, int I, uint32_t *Buf)
Definition SHA1.cpp:45
#define FP_DENORM_FLUSH_NONE
Definition SIDefines.h:1507
#define FP_DENORM_FLUSH_IN_FLUSH_OUT
Definition SIDefines.h:1504
SI Fold Operands
static void reservePrivateMemoryRegs(const TargetMachine &TM, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info)
static SDValue adjustLoadValueTypeImpl(SDValue Result, EVT LoadVT, const SDLoc &DL, SelectionDAG &DAG, bool Unpacked)
static MachineBasicBlock * emitIndirectSrc(MachineInstr &MI, MachineBasicBlock &MBB, const GCNSubtarget &ST)
static bool denormalModeIsFlushAllF64F16(const MachineFunction &MF)
static bool isAtomicRMWLegalIntTy(Type *Ty)
static void knownBitsForWorkitemID(const GCNSubtarget &ST, GISelValueTracking &VT, KnownBits &Known, unsigned Dim)
static bool flatInstrMayAccessPrivate(const Instruction *I)
Return if a flat address space atomicrmw can access private memory.
static std::pair< unsigned, int > computeIndirectRegAndOffset(const SIRegisterInfo &TRI, const TargetRegisterClass *SuperRC, unsigned VecReg, int Offset)
static bool denormalModeIsFlushAllF32(const MachineFunction &MF)
static bool addresses16Bits(int Mask)
static MachineBasicBlock * expand64BitScalarArithmetic(MachineInstr &MI, MachineBasicBlock *BB)
static bool isClampZeroToOne(SDValue A, SDValue B)
static bool supportsMin3Max3(const GCNSubtarget &Subtarget, unsigned Opc, EVT VT)
static AtomicFlushDenormalReason getAtomicFlushDenormalReason(const AtomicRMWInst *RMW)
static unsigned findFirstFreeSGPR(CCState &CCInfo)
static uint32_t getPermuteMask(SDValue V)
static SDValue lowerLaneOp(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static int getAlignedAGPRClassID(unsigned UnalignedClassID)
static void processPSInputArgs(SmallVectorImpl< ISD::InputArg > &Splits, CallingConv::ID CallConv, ArrayRef< ISD::InputArg > Ins, BitVector &Skipped, FunctionType *FType, SIMachineFunctionInfo *Info)
static SDValue selectSOffset(SDValue SOffset, SelectionDAG &DAG, const GCNSubtarget *Subtarget)
static SDValue getLoadExtOrTrunc(SelectionDAG &DAG, ISD::LoadExtType ExtType, SDValue Op, const SDLoc &SL, EVT VT)
static std::tuple< unsigned, unsigned > getDPPOpcForWaveReduction(unsigned Opc, const GCNSubtarget &ST)
static void fixMasks(SmallVectorImpl< DotSrc > &Srcs, unsigned ChainLength)
static bool is32bitWaveReduceOperation(unsigned Opc)
static TargetLowering::AtomicExpansionKind atomicSupportedIfLegalIntType(const AtomicRMWInst *RMW)
static SDValue strictFPExtFromF16(SelectionDAG &DAG, SDValue Src)
Return the source of an fp_extend from f16 to f32, or a converted FP constant.
static bool isAtomicRMWLegalXChgTy(const AtomicRMWInst *RMW)
static bool bitOpWithConstantIsReducible(unsigned Opc, uint32_t Val)
static void convertScratchAtomicToFlatAtomic(Instruction *I, unsigned PtrOpIdx)
static bool isCopyFromRegOfInlineAsm(const SDNode *N)
static bool elementPairIsOddToEven(ArrayRef< int > Mask, int Elt)
static SDValue lowerBFEIntrinsic(SDValue Op, SelectionDAG &DAG, Intrinsic::ID IntrinsicID)
static cl::opt< bool > DisableLoopAlignment("amdgpu-disable-loop-alignment", cl::desc("Do not align and prefetch loops"), cl::init(false))
static SDValue getDWordFromOffset(SelectionDAG &DAG, SDLoc SL, SDValue Src, unsigned DWordOffset)
static MachineBasicBlock::iterator loadM0FromVGPR(const SIInstrInfo *TII, MachineBasicBlock &MBB, MachineInstr &MI, unsigned InitResultReg, unsigned PhiReg, int Offset, bool UseGPRIdxMode, Register &SGPRIdxReg)
static bool isFloatingPointWaveReduceOperation(unsigned Opc)
static bool isImmConstraint(StringRef Constraint)
static SDValue padEltsToUndef(SelectionDAG &DAG, const SDLoc &DL, EVT CastVT, SDValue Src, int ExtraElts)
static bool hasCFUser(const Value *V, SmallPtrSet< const Value *, 16 > &Visited, unsigned WaveSize)
static std::pair< Register, Register > ExtractSubRegs(MachineInstr &MI, MachineOperand &Op, const TargetRegisterClass *SrcRC, const GCNSubtarget &ST, MachineRegisterInfo &MRI)
static unsigned SubIdx2Lane(unsigned Idx)
Helper function for adjustWritemask.
static TargetLowering::AtomicExpansionKind getPrivateAtomicExpansionKind(const GCNSubtarget &STI)
static bool addressMayBeAccessedAsPrivate(const MachineMemOperand *MMO, const SIMachineFunctionInfo &Info)
static MachineBasicBlock * lowerWaveReduce(MachineInstr &MI, MachineBasicBlock &BB, const GCNSubtarget &ST, unsigned Opc)
static bool elementPairIsContiguous(ArrayRef< int > Mask, int Elt)
static bool isV2BF16(Type *Ty)
static bool isFrexpExp(SDValue V, SDValue &FrexpInput)
static ArgDescriptor allocateSGPR32InputImpl(CCState &CCInfo, const TargetRegisterClass *RC, unsigned NumArgRegs)
static SDValue getMad64_32(SelectionDAG &DAG, const SDLoc &SL, EVT VT, SDValue N0, SDValue N1, SDValue N2, bool Signed)
static SDValue resolveSources(SelectionDAG &DAG, SDLoc SL, SmallVectorImpl< DotSrc > &Srcs, bool IsSigned, bool IsAny)
static bool hasNon16BitAccesses(uint64_t PermMask, SDValue &Op, SDValue &OtherOp)
static SDValue lowerWaveShuffle(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static SDValue diagnoseUnsupportedImage(SelectionDAG &DAG, SDValue Op, ArrayRef< EVT > ResultTypes, const SDLoc &DL, const Twine &Msg)
Emit a DiagnosticInfoUnsupported for an unsupported image intrinsic and return poison values of Resul...
static void placeSources(ByteProvider< SDValue > &Src0, ByteProvider< SDValue > &Src1, SmallVectorImpl< DotSrc > &Src0s, SmallVectorImpl< DotSrc > &Src1s, int Step)
static unsigned parseSyncscopeMDArg(const CallBase &CI, unsigned ArgIdx)
static EVT memVTFromLoadIntrReturn(const SITargetLowering &TLI, const DataLayout &DL, Type *Ty, unsigned MaxNumLanes)
static MachineBasicBlock::iterator emitLoadM0FromVGPRLoop(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineBasicBlock &OrigBB, MachineBasicBlock &LoopBB, const DebugLoc &DL, const MachineOperand &Idx, unsigned InitReg, unsigned ResultReg, unsigned PhiReg, unsigned InitSaveExecReg, int Offset, bool UseGPRIdxMode, Register &SGPRIdxReg)
static SDValue matchPERM(SDNode *N, TargetLowering::DAGCombinerInfo &DCI)
static bool isFrameIndexOp(SDValue Op)
static ConstantFPSDNode * getSplatConstantFP(SDValue Op)
static void allocateSGPR32Input(CCState &CCInfo, ArgDescriptor &Arg)
static void knownBitsForSBFE(const MachineInstr &MI, GISelValueTracking &VT, KnownBits &Known, const APInt &DemandedElts, unsigned BFEWidth, bool SExt, unsigned Depth)
static bool isExtendedFrom16Bits(SDValue &Operand)
static std::optional< bool > checkDot4MulSignedness(const SDValue &N, ByteProvider< SDValue > &Src0, ByteProvider< SDValue > &Src1, const SDValue &S0Op, const SDValue &S1Op, const SelectionDAG &DAG)
static bool vectorEltWillFoldAway(SDValue Op)
static SDValue getSPDenormModeValue(uint32_t SPDenormMode, SelectionDAG &DAG, const SIMachineFunctionInfo *Info, const GCNSubtarget *ST)
static uint32_t getConstantPermuteMask(uint32_t C)
static AtomicOrdering parseAtomicOrderingCABIArg(const CallBase &CI, unsigned ArgIdx)
static MachineBasicBlock * emitIndirectDst(MachineInstr &MI, MachineBasicBlock &MBB, const GCNSubtarget &ST)
static void setM0ToIndexFromSGPR(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineInstr &MI, int Offset)
static DenormalFPEnv getDenormalFPEnv(const MachineFunction &MF)
static std::pair< MachineBasicBlock *, MachineBasicBlock * > splitBlockForLoop(MachineInstr &MI, MachineBasicBlock &MBB, bool InstInLoop)
static unsigned getBasePtrIndex(const MemSDNode *N)
MemSDNode::getBasePtr() does not work for intrinsics, which needs to offset by the chain and intrinsi...
static void allocateFixedSGPRInputImpl(CCState &CCInfo, const TargetRegisterClass *RC, MCRegister Reg)
static SDValue constructRetValue(SelectionDAG &DAG, MachineSDNode *Result, ArrayRef< EVT > ResultTypes, bool IsTexFail, bool Unpacked, bool IsD16, int DMaskPop, int NumVDataDwords, bool IsAtomicPacked16Bit, const SDLoc &DL)
static std::pair< SDValue, SDValue > splitTFEValueAndStatus(SDValue Op, EVT VT, const SDLoc &DL, SelectionDAG &DAG)
static std::optional< ByteProvider< SDValue > > handleMulOperand(const SDValue &MulOperand)
static ISD::CondCode tryReduceF64CompareToHiHalf(const ISD::CondCode CC, const SDValue LHS, const SDValue RHS, const SelectionDAG &DAG)
static Register getIndirectSGPRIdx(const SIInstrInfo *TII, MachineRegisterInfo &MRI, MachineInstr &MI, int Offset)
static SDValue emitNonHSAIntrinsicError(SelectionDAG &DAG, const SDLoc &DL, EVT VT)
static EVT memVTFromLoadIntrData(const SITargetLowering &TLI, const DataLayout &DL, Type *Ty, unsigned MaxNumLanes)
static unsigned minMaxOpcToMin3Max3Opc(unsigned Opc)
static unsigned getExtOpcodeForPromotedOp(SDValue Op)
static void expand64BitV_CNDMASK(MachineInstr &MI, MachineBasicBlock *BB)
static GlobalFPAtomicLegality getGlobalMemoryFPAtomicLegality(const GCNSubtarget &Subtarget, const AtomicRMWInst *RMW, bool HasSystemScope)
static SDValue lowerBALLOTIntrinsic(const SITargetLowering &TLI, SDNode *N, SelectionDAG &DAG)
static SDValue buildSMovImm32(SelectionDAG &DAG, const SDLoc &DL, uint64_t Val)
static SDValue tryFoldMADwithSRL(SelectionDAG &DAG, const SDLoc &SL, SDValue MulLHS, SDValue MulRHS, SDValue AddRHS)
static unsigned getIntrMemWidth(unsigned IntrID)
static SDValue getBuildDwordsVector(SelectionDAG &DAG, SDLoc DL, ArrayRef< SDValue > Elts)
static SDNode * findUser(SDValue Value, unsigned Opcode)
Helper function for LowerBRCOND.
static unsigned addPermMasks(unsigned First, unsigned Second)
static uint64_t clearUnusedBits(uint64_t Val, unsigned Size)
static SDValue getFPTernOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue C, SDValue GlueChain, SDNodeFlags Flags)
static bool isV2F16OrV2BF16(Type *Ty)
static OptimizationRemark emitAtomicRMWLegalRemark(const AtomicRMWInst *RMW, GlobalFPAtomicLegality MemLegality, AtomicFlushDenormalReason DenormReason)
static SDValue emitRemovedIntrinsicError(SelectionDAG &DAG, const SDLoc &DL, EVT VT)
static SDValue getFPBinOp(SelectionDAG &DAG, unsigned Opcode, const SDLoc &SL, EVT VT, SDValue A, SDValue B, SDValue GlueChain, SDNodeFlags Flags)
static SDValue buildPCRelGlobalAddress(SelectionDAG &DAG, const GlobalValue *GV, const SDLoc &DL, int64_t Offset, EVT PtrVT, unsigned GAFlags=SIInstrInfo::MO_NONE)
static cl::opt< bool > UseDivergentRegisterIndexing("amdgpu-use-divergent-register-indexing", cl::Hidden, cl::desc("Use indirect register addressing for divergent indexes"), cl::init(false))
static const std::optional< ByteProvider< SDValue > > calculateSrcByte(const SDValue Op, uint64_t DestByte, uint64_t SrcIndex=0, unsigned Depth=0)
static bool isV2F16(Type *Ty)
static void initializeM0ToZeroForClusterLoad(SDValue Op, SelectionDAG &DAG, SDLoc DL)
static void allocateSGPR64Input(CCState &CCInfo, ArgDescriptor &Arg)
static uint64_t getIdentityValueForWaveReduction(unsigned Opc)
SI DAG Lowering interface definition.
Interface definition for SIRegisterInfo.
const char * Msg
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
#define LLVM_DEBUG(...)
Definition Debug.h:119
static unsigned getScalarSizeInBits(Type *Ty)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
LLVM IR instance of the generic uniformity analysis.
static constexpr int Concat[]
Value * RHS
Value * LHS
The Input class is used to parse a yaml document into in-memory structs and vectors.
static std::optional< uint32_t > getLDSKernelIdMetadata(const Function &F)
void setDynLDSAlign(const Function &F, const GlobalVariable &GV)
static std::optional< uint32_t > get32BitAbsoluteAddress(const GlobalValue &GV, unsigned AS)
static unsigned numBitsSigned(SDValue Op, SelectionDAG &DAG)
SDValue SplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Split a vector load into 2 loads of half the vector.
void analyzeFormalArgumentsCompute(CCState &State, const SmallVectorImpl< ISD::InputArg > &Ins) const
The SelectionDAGBuilder will automatically promote function arguments with illegal types.
SDValue LowerF64ToF16Safe(SDValue Src, const SDLoc &DL, SelectionDAG &DAG) const
SDValue storeStackInputValue(SelectionDAG &DAG, const SDLoc &SL, SDValue Chain, SDValue ArgVal, int64_t Offset) const
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
SDValue splitBinaryBitConstantOpImpl(DAGCombinerInfo &DCI, const SDLoc &SL, unsigned Opc, SDValue LHS, uint32_t ValLo, uint32_t ValHi) const
Split the 64-bit value LHS into two 32-bit components, and perform the binary operation Opc to it wit...
SDValue lowerUnhandledCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals, StringRef Reason) const
virtual SDValue LowerGlobalAddress(AMDGPUMachineFunctionInfo *MFI, SDValue Op, SelectionDAG &DAG) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
SDValue addTokenForArgument(SDValue Chain, SelectionDAG &DAG, MachineFrameInfo &MFI, int ClobberedFI) const
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
static bool needsDenormHandlingF32(const SelectionDAG &DAG, SDValue Src, SDNodeFlags Flags)
uint32_t getImplicitParameterOffset(const MachineFunction &MF, const ImplicitParameter Param) const
Helper function that returns the byte offset of the given type of implicit parameter.
SDValue LowerFP_TO_INT(SDValue Op, SelectionDAG &DAG) const
SDValue loadInputValue(SelectionDAG &DAG, const TargetRegisterClass *RC, EVT VT, const SDLoc &SL, const ArgDescriptor &Arg) const
AMDGPUTargetLowering(const TargetMachine &TM, const TargetSubtargetInfo &STI, const AMDGPUSubtarget &AMDGPUSTI)
static EVT getEquivalentMemType(LLVMContext &Context, EVT VT)
SDValue LowerBlockAddress(SDValue Op, SelectionDAG &DAG) const
SDValue CreateLiveInRegister(SelectionDAG &DAG, const TargetRegisterClass *RC, Register Reg, EVT VT, const SDLoc &SL, bool RawReg=false) const
Helper function that adds Reg to the LiveIn list of the DAG's MachineFunction.
SDValue SplitVectorStore(SDValue Op, SelectionDAG &DAG) const
Split a vector store into 2 stores of half the vector.
std::pair< SDValue, SDValue > split64BitValue(SDValue Op, SelectionDAG &DAG) const
Return 64-bit value Op as two 32-bit integers.
static CCAssignFn * CCAssignFnForReturn(CallingConv::ID CC, bool IsVarArg)
static CCAssignFn * CCAssignFnForCall(CallingConv::ID CC, bool IsVarArg)
Selects the correct CCAssignFn for a given CallingConvention value.
static unsigned numBitsUnsigned(SDValue Op, SelectionDAG &DAG)
static bool allowApproxFunc(const SelectionDAG &DAG, SDNodeFlags Flags)
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
SDValue performRcpCombine(SDNode *N, DAGCombinerInfo &DCI) const
SDValue LowerFP_TO_INT_SAT(SDValue Op, SelectionDAG &DAG) const
static bool shouldFoldFNegIntoSrc(SDNode *FNeg, SDValue FNegSrc)
bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const override
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
SDValue WidenOrSplitVectorLoad(SDValue Op, SelectionDAG &DAG) const
Widen a suitably aligned v3 load.
SDValue getHiHalf64(SDValue Op, SelectionDAG &DAG) const
bool isNoopAddrSpaceCast(const DataLayout &DL, unsigned SrcAS, unsigned DestAS) const override
Returns true if a cast between SrcAS and DestAS is a noop.
const std::array< unsigned, 3 > & getDims() const
static const LaneMaskConstants & get(const GCNSubtarget &ST)
static const fltSemantics & IEEEsingle()
Definition APFloat.h:304
static constexpr roundingMode rmNearestTiesToEven
Definition APFloat.h:361
static const fltSemantics & IEEEhalf()
Definition APFloat.h:302
static APFloat getQNaN(const fltSemantics &Sem, bool Negative=false, const APInt *payload=nullptr)
Factory for QNaN values.
Definition APFloat.h:1224
LLVM_ABI opStatus convert(const fltSemantics &ToSemantics, roundingMode RM, bool *losesInfo)
Definition APFloat.cpp:6034
LLVM_READONLY int getExactLog2Abs() const
Definition APFloat.h:1639
bool isNegative() const
Definition APFloat.h:1583
bool isNormal() const
Definition APFloat.h:1587
static APFloat getOne(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative One.
Definition APFloat.h:1192
static APFloat getLargest(const fltSemantics &Sem, bool Negative=false)
Returns the largest finite number in the given semantics.
Definition APFloat.h:1242
static APFloat getInf(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Infinity.
Definition APFloat.h:1202
static APFloat getZero(const fltSemantics &Sem, bool Negative=false)
Factory for Positive and Negative Zero.
Definition APFloat.h:1183
bool isInfinity() const
Definition APFloat.h:1580
Class for arbitrary precision integers.
Definition APInt.h:78
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
static APInt getMaxValue(unsigned numBits)
Gets maximum unsigned value of APInt for specific bit width.
Definition APInt.h:202
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
bool isZero() const
Determine if this value is zero, i.e. all bits are clear.
Definition APInt.h:376
bool isSignMask() const
Check if the APInt's value is returned by getSignMask.
Definition APInt.h:462
unsigned countr_zero() const
Count the number of trailing zero bits.
Definition APInt.h:1659
bool isOneBitSet(unsigned BitNo) const
Determine if this APInt Value only has the specified bit set.
Definition APInt.h:362
bool isSignBitSet() const
Determine if sign bit of this APInt is set.
Definition APInt.h:337
static APInt getHighBitsSet(unsigned numBits, unsigned hiBitsSet)
Constructs an APInt value that has the top hiBitsSet bits set.
Definition APInt.h:292
bool sge(const APInt &RHS) const
Signed greater or equal comparison.
Definition APInt.h:1241
bool uge(const APInt &RHS) const
Unsigned greater or equal comparison.
Definition APInt.h:1225
This class represents an incoming formal argument to a Function.
Definition Argument.h:32
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Check if an argument has a given attribute.
Definition Function.cpp:336
const Function * getParent() const
Definition Argument.h:44
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
bool empty() const
Check if the array is empty.
Definition ArrayRef.h:136
An instruction that atomically checks whether a specified value is in a memory location,...
bool isVolatile() const
Return true if this is a cmpxchg from a volatile memory location.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
Align getAlign() const
Return the alignment of the memory that is being allocated by the instruction.
static unsigned getPointerOperandIndex()
an instruction that atomically reads a memory location, combines it with another value,...
static unsigned getPointerOperandIndex()
BinOp
This enumeration lists the possible modifications atomicrmw can make.
@ Add
*p = old + v
@ FAdd
*p = old + v
@ USubCond
Subtract only if no unsigned overflow.
@ Min
*p = old <signed v ? old : v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
@ USubSat
*p = usub.sat(old, v) usub.sat matches the behavior of llvm.usub.sat.
@ FSub
*p = old - v
@ UIncWrap
Increment one up to a maximum value.
@ Max
*p = old >signed v ? old : v
@ UMin
*p = old <unsigned v ? old : v
@ FMin
*p = minnum(old, v) minnum matches the behavior of llvm.minnum.
@ UMax
*p = old >unsigned v ? old : v
@ FMax
*p = maxnum(old, v) maxnum matches the behavior of llvm.maxnum.
@ UDecWrap
Decrement one until a minimum value or zero.
@ Nand
*p = ~(old & v)
void setOperation(BinOp Operation)
BinOp getOperation() const
SyncScope::ID getSyncScopeID() const
Returns the synchronization scope ID of this rmw instruction.
static LLVM_ABI StringRef getOperationName(BinOp Op)
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
bool isCompareAndSwap() const
Returns true if this SDNode represents cmpxchg atomic operation, false otherwise.
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
LLVM_ABI MemoryEffects getMemoryEffects() const
LLVM Basic Block Representation.
Definition BasicBlock.h:62
iterator end()
Definition BasicBlock.h:459
LLVM_ABI BasicBlock * splitBasicBlock(iterator I, const Twine &BBName="")
Split the basic block into two basic blocks at the specified instruction.
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
static BasicBlock * Create(LLVMContext &Context, const Twine &Name="", Function *Parent=nullptr, BasicBlock *InsertBefore=nullptr)
Creates a new BasicBlock.
Definition BasicBlock.h:206
A "pseudo-class" with methods for operating on BUILD_VECTORs.
Represents known origin of an individual byte in combine pattern.
static ByteProvider getConstantZero()
static ByteProvider getSrc(std::optional< ISelOp > Val, int64_t ByteOffset, int64_t VectorOffset)
CCState - This class holds information needed while lowering arguments and return values.
MachineFunction & getMachineFunction() const
unsigned getFirstUnallocated(ArrayRef< MCPhysReg > Regs) const
getFirstUnallocated - Return the index of the first unallocated register in the set,...
static LLVM_ABI bool resultsCompatible(CallingConv::ID CalleeCC, CallingConv::ID CallerCC, MachineFunction &MF, LLVMContext &C, const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn CalleeFn, CCAssignFn CallerFn)
Returns true if the results of the two calling conventions are compatible.
LLVM_ABI void AnalyzeCallResult(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeCallResult - Analyze the return values of a call, incorporating info about the passed values i...
MCRegister AllocateReg(MCPhysReg Reg)
AllocateReg - Attempt to allocate one register.
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
int64_t AllocateStack(unsigned Size, Align Alignment)
AllocateStack - Allocate a chunk of stack space with the specified size and alignment.
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
bool isAllocated(MCRegister Reg) const
isAllocated - Return true if the specified register (or an alias) is allocated.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
int64_t getLocMemOffset() const
Base class for all callable instructions (InvokeInst and CallInst) Holds everything related to callin...
bool hasFnAttr(Attribute::AttrKind Kind) const
Determine whether this call has the given attribute.
LLVM_ABI bool isMustTailCall() const
Tests if this call site must be tail call optimized.
Value * getArgOperand(unsigned i) const
unsigned arg_size() const
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
static LLVM_ABI CastInst * CreatePointerCast(Value *S, Type *Ty, const Twine &Name="", InsertPosition InsertBefore=nullptr)
Create a BitCast, AddrSpaceCast or a PtrToInt cast instruction.
const APFloat & getValueAPF() const
bool isPosZero() const
Return true if the value is positive zero.
bool isOne() const
Returns true if this value is exactly +1.0.
bool isMinusOne() const
Returns true if this value is exactly -1.0.
bool isNegative() const
Return true if the value is negative.
bool isInfinity() const
Return true if the value is an infinity.
This is the shared class of boolean and integer constants.
Definition Constants.h:87
bool isZero() const
This is just a convenience method to make client code smaller for a common code.
Definition Constants.h:219
uint64_t getZExtValue() const
const APInt & getAPIntValue() const
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
LLVM_ABI Align getABITypeAlign(Type *Ty) const
Returns the minimum ABI-required alignment for the specified type.
bool isBigEndian() const
Definition DataLayout.h:218
A debug info location.
Definition DebugLoc.h:126
Diagnostic information for unsupported feature in backend.
static constexpr ElementCount getFixed(ScalarTy MinVal)
Definition TypeSize.h:305
Class to represent fixed width SIMD vectors.
unsigned getNumElements() const
FunctionLoweringInfo - This contains information that is global to a function that is used when lower...
Register DemoteRegister
DemoteRegister - if CanLowerReturn is false, DemoteRegister is a vreg allocated to hold a pointer to ...
LLVM_ABI const Value * getValueFromVirtualReg(Register Vreg)
This method is called from TargetLowerinInfo::isSDNodeSourceOfDivergence to get the Value correspondi...
Class to represent function types.
Type * getParamType(unsigned i) const
Parameter type accessors.
FunctionType * getFunctionType() const
Returns the FunctionType for me.
Definition Function.h:212
const DataLayout & getDataLayout() const
Get the data layout of the module this function belongs to.
Definition Function.cpp:360
iterator_range< arg_iterator > args()
Definition Function.h:877
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
LLVMContext & getContext() const
getContext - Return a reference to the LLVMContext associated with this function.
Definition Function.cpp:356
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
Definition Function.cpp:810
size_t arg_size() const
Definition Function.h:886
Argument * getArg(unsigned i) const
Definition Function.h:871
const SIInstrInfo * getInstrInfo() const override
bool hasMadF16() const
unsigned getInstCacheLineSize() const
Instruction cache line size in bytes (64 for pre-GFX11, 128 for GFX11+).
const SIRegisterInfo * getRegisterInfo() const override
bool hasMin3Max3_16() const
bool supportsWaveWideBPermute() const
unsigned getMaxPrivateElementSize(bool ForBufferRSrc=false) const
bool isWave64() const
bool hasPrivateSegmentBuffer() const
const MachineFunction & getMachineFunction() const
void computeKnownBitsImpl(Register R, KnownBits &Known, const APInt &DemandedElts, unsigned Depth=0)
Evaluate a known-bits query with an explicit worklist instead of recursive descent.
bool isDivergentAtDef(ConstValueRefT V) const
Whether V is divergent at its definition.
LLVM_ABI unsigned getAddressSpace() const
const GlobalValue * getGlobal() const
bool hasExternalLinkage() const
unsigned getAddressSpace() const
Module * getParent()
Get the module that this global value is contained inside of...
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this global belongs to.
Definition Globals.cpp:205
Type * getValueType() const
LLVM_ABI uint64_t getGlobalSize(const DataLayout &DL) const
Get the size of this global variable in bytes.
Definition Globals.cpp:640
This provides a uniform API for creating instructions and inserting them into a basic block: either a...
Definition IRBuilder.h:2918
LLVM_ABI Instruction * clone() const
Create a copy of 'this' instruction that is identical in all ways except the following:
LLVM_ABI void removeFromParent()
This method unlinks 'this' from the containing basic block, but does not delete it.
bool hasMetadata() const
Return true if this instruction has any metadata attached to it.
LLVM_ABI const Function * getFunction() const
Return the function this instruction belongs to.
LLVM_ABI void setMetadata(unsigned KindID, MDNode *Node)
Set the metadata of the specified kind to the specified node.
LLVM_ABI const DataLayout & getDataLayout() const
Get the data layout of the module this instruction belongs to.
LLVM_ABI InstListType::iterator insertInto(BasicBlock *ParentBB, InstListType::iterator It)
Inserts an unlinked instruction into ParentBB at position It and returns the iterator of the inserted...
Class to represent integer types.
A wrapper class for inspecting calls to intrinsic functions.
constexpr unsigned getScalarSizeInBits() const
static constexpr LLT scalar(unsigned SizeInBits)
Get a low-level scalar or aggregate "bag of bits".
static constexpr LLT pointer(unsigned AddressSpace, unsigned SizeInBits)
Get a low-level pointer in the given address space.
constexpr TypeSize getSizeInBits() const
Returns the total size of the type. Must only be called on sized types.
static LLT integer(unsigned SizeInBits)
LLT changeElementSize(unsigned NewEltSize) const
If this type is a vector, return a vector with the same number of elements but the new element size.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void emitError(const Instruction *I, const Twine &ErrorStr)
emitError - Emit an error message to the currently installed error handler with optional location inf...
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
LLVM_ABI SyncScope::ID getOrInsertSyncScopeID(StringRef SSN)
getOrInsertSyncScopeID - Maps synchronization scope name to synchronization scope ID.
An instruction for reading from memory.
unsigned getPointerAddressSpace() const
Returns the address space of the pointer operand.
static unsigned getPointerOperandIndex()
This class is used to represent ISD::LOAD nodes.
bool hasValue() const
TypeSize getValue() const
Describe properties that are true of each instruction in the target description file.
unsigned getID() const
getID() - Return the register class ID number.
MCRegister getRegister(unsigned i) const
getRegister - Return the specified register in the class.
unsigned getNumRegs() const
getNumRegs - Return the number of registers in this class.
iterator begin() const
begin/end - Return all of the registers in this class.
Wrapper class representing physical registers. Should be passed by value.
Definition MCRegister.h:41
LLVM_ABI MDNode * createRange(const APInt &Lo, const APInt &Hi)
Return metadata describing the range [Lo, Hi).
Definition MDBuilder.cpp:96
Metadata node.
Definition Metadata.h:1081
const MDOperand & getOperand(unsigned I) const
Definition Metadata.h:1437
Helper class for constructing bundles of MachineInstrs.
MachineBasicBlock::instr_iterator begin() const
Return an iterator to the first bundled instruction.
Machine Value Type.
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
bool bitsLE(MVT VT) const
Return true if this has no more bits than VT.
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool isScalableVector() const
Return true if this is a vector value type where the runtime length is machine dependent.
static LLVM_ABI MVT getVT(Type *Ty, bool HandleUnknown=false)
Return the value type corresponding to the specified type.
static auto all_valuetypes()
SimpleValueType Iteration.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
static MVT getVectorVT(MVT VT, unsigned NumElements)
static MVT getIntegerVT(unsigned BitWidth)
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI iterator getFirstTerminator()
Returns an iterator to the first terminator instruction of this basic block.
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
LLVM_ABI MachineBasicBlock * splitAt(MachineInstr &SplitInst, bool UpdateLiveIns=true, LiveIntervals *LIS=nullptr)
Split a basic block into 2 pieces at SplitPoint.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
Align getAlignment() const
Return alignment of the basic block.
MachineInstrBundleIterator< MachineInstr > iterator
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
bool hasCalls() const
Return true if the current function has any function calls.
void setHasTailCall(bool V=true)
void setReturnAddressIsTaken(bool s)
bool hasStackObjects() const
Return true if there are any stack objects in this function.
PseudoSourceValueManager & getPSVManager() const
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
DenormalMode getDenormalMode(const fltSemantics &FPType) const
Returns the denormal handling type for the default rounding mode of the function.
void push_back(MachineBasicBlock *MBB)
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & addDef(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a virtual register definition operand.
const MachineInstrBuilder & cloneMemRefs(const MachineInstr &OtherMI) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
Representation of each machine instruction.
bool isMoveImmediate(QueryType Type=IgnoreBundle) const
Return true if this instruction is a move immediate (including conditional moves) instruction.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
LocationSize getSize() const
Return the size in bytes of the memory reference.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MONonTemporal
The memory access is non-temporal.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
Flags getFlags() const
Return the raw flags of the source value,.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
static MachineOperand CreateImm(int64_t Val)
void setIsUndef(bool Val=true)
Register getReg() const
getReg - Returns the register number.
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI bool hasOneNonDBGUse(Register RegNo) const
hasOneNonDBGUse - Return true if there is exactly one non-Debug use of the specified register.
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI void clearKillFlags(Register Reg) const
clearKillFlags - Iterate over all the uses of the given register and clear the kill flag from the Mac...
LLVM_ABI LLVM_READONLY MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
LLT getType(Register Reg) const
Get the low-level type of Reg or LLT{} if Reg is not a generic (target independent) virtual register.
LLVM_ABI bool isLiveIn(Register Reg) const
LLVM_ABI void setType(Register VReg, LLT Ty)
Set the low-level type of VReg to Ty.
LLVM_ABI void setRegClass(Register Reg, const TargetRegisterClass *RC)
setRegClass - Set the register class of the specified virtual register.
LLVM_ABI Register getLiveInVirtReg(MCRegister PReg) const
getLiveInVirtReg - If PReg is a live-in physical register, return the corresponding live-in virtual r...
const TargetRegisterClass * getRegClassOrNull(Register Reg) const
Return the register class of Reg, or null if Reg has not been assigned a register class yet.
void setSimpleHint(Register VReg, Register PrefReg)
Specify the preferred (target independent) register allocation hint for the specified virtual registe...
LLVM_ABI Register cloneVirtualRegister(Register VReg, StringRef Name="")
Create and return a new virtual register in the function with the same attributes as the given regist...
unsigned getNumVirtRegs() const
getNumVirtRegs - Return the number of virtual registers created.
LLVM_ABI void replaceRegWith(Register FromReg, Register ToReg)
replaceRegWith - Replace all instances of FromReg with ToReg in the machine function.
An SDNode that represents everything that will be needed to construct a MachineInstr.
This is an abstract virtual class for memory operations.
unsigned getAddressSpace() const
Return the address space for the associated pointer.
MachineMemOperand * getMemOperand() const
Return the unique MachineMemOperand object describing the memory reference performed by operation.
EVT getMemoryVT() const
Return the type of the in-memory value.
bool onlyWritesMemory() const
Whether this function only (at most) writes memory.
Definition ModRef.h:252
bool doesNotAccessMemory() const
Whether this function accesses no memory.
Definition ModRef.h:246
bool onlyReadsMemory() const
Whether this function only (at most) reads memory.
Definition ModRef.h:249
const Triple & getTargetTriple() const
Get the target triple which is a string describing the target host.
Definition Module.h:328
The optimization diagnostic interface.
LLVM_ABI void emit(DiagnosticInfoOptimizationBase &OptDiag)
Output the remark via the diagnostic handler and to the optimization record file.
Diagnostic information for applied optimization remarks.
static LLVM_ABI PointerType * get(LLVMContext &C, unsigned AddressSpace)
This constructs an opaque pointer to an object in a numbered address space.
Definition Type.cpp:887
static LLVM_ABI PoisonValue * get(Type *T)
Static factory methods - Return an 'poison' object of the specified type.
LLVM_ABI const PseudoSourceValue * getConstantPool()
Return a pseudo source value referencing the constant pool.
Wrapper class representing virtual and physical registers.
Definition Register.h:20
static Register index2VirtReg(unsigned Index)
Convert a 0-based index to a virtual register number.
Definition Register.h:72
constexpr bool isPhysical() const
Return true if the specified register number is in the physical register namespace.
Definition Register.h:83
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool isDivergent() const
bool hasOneUse() const
Return true if there is exactly one use of this node.
value_iterator value_end() const
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumValues() const
Return the number of values defined/returned by this operator.
const SDValue & getOperand(unsigned Num) const
uint64_t getConstantOperandVal(unsigned Num) const
Helper method returns the integer value of a ConstantSDNode operand.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
user_iterator user_begin() const
Provide iteration support to walk over all users of an SDNode.
op_iterator op_end() const
bool isAnyAdd() const
Returns true if the node type is ADD or PTRADD.
value_iterator value_begin() const
op_iterator op_begin() const
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
unsigned getNumOperands() const
static unsigned getMaxMUBUFImmOffset(const GCNSubtarget &ST)
static unsigned getDSShaderTypeValue(const MachineFunction &MF)
This class keeps track of the SPI_SP_INPUT_ADDR config register, which tells the hardware which inter...
AMDGPU::ClusterDimsAttr getClusterDims() const
SIModeRegisterDefaults getMode() const
std::tuple< const ArgDescriptor *, const TargetRegisterClass *, LLT > getPreloadedValue(AMDGPUFunctionArgInfo::PreloadedValue Value) const
const AMDGPUGWSResourcePseudoSourceValue * getGWSPSV(const AMDGPUTargetMachine &TM)
static unsigned getSubRegFromChannel(unsigned Channel, unsigned NumRegs=1)
static LLVM_READONLY const TargetRegisterClass * getSGPRClassForBitWidth(unsigned BitWidth)
static bool isVGPRClass(const TargetRegisterClass *RC)
static bool isSGPRClass(const TargetRegisterClass *RC)
static bool isAGPRClass(const TargetRegisterClass *RC)
bool isOffsetFoldingLegal(const GlobalAddressSDNode *GA) const override
Return true if folding a constant offset with the given GlobalAddress is legal.
bool isTypeDesirableForOp(unsigned Op, EVT VT) const override
Return true if the target has native support for the specified value type and it is 'desirable' to us...
SDNode * PostISelFolding(MachineSDNode *N, SelectionDAG &DAG) const override
Fold the instructions after selecting them.
SDValue splitTernaryVectorOp(SDValue Op, SelectionDAG &DAG) const
MachineSDNode * wrapAddr64Rsrc(SelectionDAG &DAG, const SDLoc &DL, SDValue Ptr) const
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
SDValue lowerGET_ROUNDING(SDValue Op, SelectionDAG &DAG) const
AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
bool requiresUniformRegister(MachineFunction &MF, const Value *V) const override
Allows target to decide about the register class of the specific value that is live outside the defin...
bool isFMADLegal(const SelectionDAG &DAG, const SDNode *N) const override
Returns true if be combined with to form an ISD::FMAD.
AtomicExpansionKind shouldExpandAtomicStoreInIR(StoreInst *SI) const override
Returns how the given (atomic) store should be expanded by the IR-level AtomicExpand pass into.
void bundleInstWithWaitcnt(MachineInstr &MI) const
Insert MI into a BUNDLE with an S_WAITCNT 0 immediately following it.
SDValue lowerROTR(SDValue Op, SelectionDAG &DAG) const
MVT getScalarShiftAmountTy(const DataLayout &, EVT) const override
Return the type to use for a scalar shift opcode, given the shifted amount type.
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
MVT getPointerTy(const DataLayout &DL, unsigned AS) const override
Map address space 7 to MVT::amdgpuBufferFatPointer because that's its in-memory representation.
bool denormalsEnabledForType(const SelectionDAG &DAG, EVT VT) const
void insertCopiesSplitCSR(MachineBasicBlock *Entry, const SmallVectorImpl< MachineBasicBlock * > &Exits) const override
Insert explicit copies in entry and exit blocks.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
Return the ValueType of the result of SETCC operations.
SDNode * legalizeTargetIndependentNode(SDNode *Node, SelectionDAG &DAG) const
Legalize target independent instructions (e.g.
bool allowsMisalignedMemoryAccessesImpl(unsigned Size, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *IsFast=nullptr) const
TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const override
Return the preferred vector type legalization action.
SDValue lowerFP_EXTEND(SDValue Op, SelectionDAG &DAG) const
const GCNSubtarget * getSubtarget() const
bool enableAggressiveFMAFusion(EVT VT) const override
Return true if target always benefits from combining into FMA for a given value type.
bool shouldEmitGOTReloc(const GlobalValue *GV) const
SDValue splitUnaryVectorOp(SDValue Op, SelectionDAG &DAG) const
SDValue lowerGET_FPENV(SDValue Op, SelectionDAG &DAG) const
bool isCanonicalized(SelectionDAG &DAG, SDValue Op, SDNodeFlags UserFlags={}, unsigned MaxDepth=5) const
void allocateSpecialInputSGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
void allocateLDSKernelId(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
SDValue LowerSTACKSAVE(SDValue Op, SelectionDAG &DAG) const
bool isReassocProfitable(SelectionDAG &DAG, SDValue N0, SDValue N1) const override
void allocateHSAUserSGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
ArrayRef< MCPhysReg > getRoundingControlRegisters() const override
Returns a 0 terminated array of rounding control registers that can be attached into strict FP call.
ConstraintType getConstraintType(StringRef Constraint) const override
Given a constraint, return the type of constraint it is for this target.
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent) const override
Return the register class that should be used for the specified value type.
void AddMemOpInit(MachineInstr &MI) const
MachineMemOperand::Flags getTargetMMOFlags(const Instruction &I) const override
This callback is used to inspect load/store instructions and add target-specific MachineMemOperand fl...
bool isLegalGlobalAddressingMode(const AddrMode &AM) const
bool shouldConvertConstantLoadToIntImm(const APInt &Imm, Type *Ty) const override
Return true if it is beneficial to convert a load of a constant to just the constant itself.
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
void emitExpandAtomicStore(StoreInst *SI) const override
Perform a atomic store using a target-specific way.
AtomicExpansionKind shouldExpandAtomicLoadInIR(LoadInst *LI) const override
Returns how the given (atomic) load should be expanded by the IR-level AtomicExpand pass.
Align computeKnownAlignForTargetInstr(GISelValueTracking &Analysis, Register R, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
Determine the known alignment for the pointer value R.
bool getAsmOperandConstVal(SDValue Op, uint64_t &Val) const
bool isShuffleMaskLegal(ArrayRef< int >, EVT) const override
Targets can use this to indicate that they only support some VECTOR_SHUFFLE operations,...
void emitExpandAtomicLoad(LoadInst *LI) const override
Perform a atomic load using a target-specific way.
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
Returns the target specific optimal type for load and store operations as a result of memset,...
void computeKnownBitsForStackObjectPointer(KnownBits &Known, const MachineFunction &MF, Align Alignment) const override
Determine known bits of a pointer to a known valid stack object.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const override
Lower the specified operand into the Ops vector.
LLT getPreferredShiftAmountTy(LLT Ty) const override
Return the preferred type to use for a shift opcode, given the shifted amount type is ShiftValueTy.
ExtractSubvectorCost getExtractSubvectorCost(EVT ResVT, EVT SrcVT, unsigned Index) const override
Return the cost of extracting a subvector of type ResVT from a vector of type SrcVT,...
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
Align getPrefLoopAlignment(MachineLoop *ML, const MachineBasicBlock *BlockToAlign) const override
Return the preferred loop alignment.
SDValue lowerSET_FPENV(SDValue Op, SelectionDAG &DAG) const
bool shouldPreservePtrArith(const Function &F, EVT PtrVT) const override
True if target has some particular form of dealing with pointer arithmetic semantics for pointers wit...
void getTgtMemIntrinsic(SmallVectorImpl< IntrinsicInfo > &, const CallBase &, MachineFunction &MF, unsigned IntrinsicID) const override
Given an intrinsic, checks if on the target the intrinsic will need to map to a MemIntrinsicNode (tou...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
SDValue lowerSET_ROUNDING(SDValue Op, SelectionDAG &DAG) const
void allocateSpecialInputVGPRsFixed(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
Allocate implicit function VGPR arguments in fixed registers.
MachineBasicBlock * emitGWSMemViolTestLoop(MachineInstr &MI, MachineBasicBlock *BB) const
bool getAddrModeArguments(const IntrinsicInst *I, SmallVectorImpl< Value * > &Ops, Type *&AccessTy) const override
CodeGenPrepare sinks address calculations into the same BB as Load/Store instructions reading the add...
bool checkAsmConstraintValA(SDValue Op, uint64_t Val, unsigned MaxSize=64) const
bool shouldEmitFixup(const GlobalValue *GV) const
MachineBasicBlock * splitKillBlock(MachineInstr &MI, MachineBasicBlock *BB) const
void emitExpandAtomicCmpXchg(AtomicCmpXchgInst *CI) const override
Perform a cmpxchg expansion using a target-specific method.
bool canTransformPtrArithOutOfBounds(const Function &F, EVT PtrVT) const override
True if the target allows transformations of in-bounds pointer arithmetic that cause out-of-bounds in...
bool hasMemSDNodeUser(SDNode *N) const
bool isSDNodeSourceOfDivergence(const SDNode *N, FunctionLoweringInfo *FLI, UniformityInfo *UA) const override
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
bool isEligibleForTailCallOptimization(SDValue Callee, CallingConv::ID CalleeCC, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SmallVectorImpl< ISD::InputArg > &Ins, SelectionDAG &DAG) const
bool isMemOpHasNoClobberedMemOperand(const SDNode *N) const
bool isLegalFlatAddressingMode(const AddrMode &AM, unsigned AddrSpace) const
SDValue LowerCallResult(SDValue Chain, SDValue InGlue, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals, bool isThisReturn, SDValue ThisVal) const
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
bool isFPExtFoldable(const SelectionDAG &DAG, unsigned Opcode, EVT DestVT, EVT SrcVT) const override
Return true if an fpext operation input to an Opcode operation is free (for instance,...
void AdjustInstrPostInstrSelection(MachineInstr &MI, SDNode *Node) const override
Assign the register class depending on the number of bits set in the writemask.
MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void finalizeLowering(MachineFunction &MF) const override
Execute target specific actions to finalize target lowering.
static bool isNonGlobalAddrSpace(unsigned AS)
void emitExpandAtomicAddrSpacePredicate(Instruction *AI) const
MachineSDNode * buildRSRC(SelectionDAG &DAG, const SDLoc &DL, SDValue Ptr, uint32_t RsrcDword1, uint64_t RsrcDword2And3) const
Return a resource descriptor with the 'Add TID' bit enabled The TID (Thread ID) is multiplied by the ...
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
bool mayBeEmittedAsTailCall(const CallInst *) const override
Return true if the target may be able emit the call instruction as a tail call.
void passSpecialInputs(CallLoweringInfo &CLI, CCState &CCInfo, const SIMachineFunctionInfo &Info, SmallVectorImpl< std::pair< unsigned, SDValue > > &RegsToPass, SmallVectorImpl< SDValue > &MemOpChains, SDValue Chain) const
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
bool checkAsmConstraintVal(SDValue Op, StringRef Constraint, uint64_t Val) const
bool isKnownNeverNaNForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, bool SNaN=false, unsigned Depth=0) const override
If SNaN is false,.
void emitExpandAtomicRMW(AtomicRMWInst *AI) const override
Perform a atomicrmw expansion using a target-specific way.
static bool shouldExpandVectorDynExt(unsigned EltSize, unsigned NumElem, bool IsDivergentIdx, const GCNSubtarget *Subtarget)
Check if EXTRACT_VECTOR_ELT/INSERT_VECTOR_ELT (<n x e>, var-idx) should be expanded into a set of cmp...
bool shouldUseLDSConstAddress(const GlobalValue *GV) const
bool supportSplitCSR(MachineFunction *MF) const override
Return true if the target supports that a subset of CSRs for the given machine function is handled ex...
bool isExtractVecEltCheap(EVT VT, unsigned Index) const override
Return true if extraction of a scalar element from the given vector type at the given index is cheap.
SDValue LowerDYNAMIC_STACKALLOC(SDValue Op, SelectionDAG &DAG) const
bool allowsMisalignedMemoryAccesses(LLT Ty, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *IsFast=nullptr) const override
LLT handling variant.
bool canMergeStoresTo(unsigned AS, EVT MemVT, const MachineFunction &MF) const override
Returns if it's reasonable to merge stores to MemVT size.
SDValue lowerPREFETCH(SDValue Op, SelectionDAG &DAG) const
SITargetLowering(const TargetMachine &tm, const GCNSubtarget &STI)
void computeKnownBitsForTargetInstr(GISelValueTracking &Analysis, Register R, KnownBits &Known, const APInt &DemandedElts, const MachineRegisterInfo &MRI, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
bool isFreeAddrSpaceCast(const DataLayout &DL, unsigned SrcAS, unsigned DestAS) const override
Returns true if a cast from SrcAS to DestAS is "cheap", such that e.g.
bool shouldEmitPCReloc(const GlobalValue *GV) const
bool isUniformLoad(const LoadSDNode *Load) const
AtomicExpansionKind shouldExpandAtomicCmpXchgInIR(const AtomicCmpXchgInst *AI) const override
Returns how the given atomic cmpxchg should be expanded by the IR-level AtomicExpand pass.
void initializeSplitCSR(MachineBasicBlock *Entry) const override
Perform necessary initialization to handle a subset of CSRs explicitly via copies.
void allocateSpecialEntryInputVGPRs(CCState &CCInfo, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
void allocatePreloadKernArgSGPRs(CCState &CCInfo, SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< ISD::InputArg > &Ins, MachineFunction &MF, const SIRegisterInfo &TRI, SIMachineFunctionInfo &Info) const
SDValue copyToM0(SelectionDAG &DAG, SDValue Chain, const SDLoc &DL, SDValue V) const
SDValue splitBinaryVectorOp(SDValue Op, SelectionDAG &DAG) const
MachinePointerInfo getKernargSegmentPtrInfo(MachineFunction &MF) const
unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
MVT getPointerMemTy(const DataLayout &DL, unsigned AS) const override
Similarly, the in-memory representation of a p7 is {p8, i32}, aka v8i32 when padding is added.
void allocateSystemSGPRs(CCState &CCInfo, MachineFunction &MF, SIMachineFunctionInfo &Info, CallingConv::ID CallConv, bool IsShader) const
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
unsigned getMaxPermittedBytesForAlignment(MachineBasicBlock *MBB) const override
Return the maximum amount of bytes allowed to be emitted when padding for alignment.
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
SDValue getExtractVectorElt(const SDLoc &DL, EVT VT, SDValue Vec, unsigned Idx)
Extract element at Idx from Vec.
const SDValue & getRoot() const
Return the root tag of the SelectionDAG.
bool isKnownNeverSNaN(SDValue Op, const APInt &DemandedElts, unsigned Depth=0) const
const TargetSubtargetInfo & getSubtarget() const
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getShiftAmountConstant(uint64_t Val, EVT VT, const SDLoc &DL)
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI void ExtractVectorElements(SDValue Op, SmallVectorImpl< SDValue > &Args, unsigned Start=0, unsigned Count=0, EVT EltVT=EVT())
Append the extracted elements from Start to Count out of the vector Op in Args.
LLVM_ABI SDValue getAtomicLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT MemVT, EVT VT, SDValue Chain, SDValue Ptr, MachineMemOperand *MMO)
LLVM_ABI SDValue getFreeze(SDValue V)
Return a freeze using the SDLoc of the value operand.
LLVM_ABI bool isConstantIntBuildVectorOrConstantInt(SDValue N, bool AllowOpaques=true) const
Test whether the given value is a constant int or similar node.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getConstantFP(double Val, const SDLoc &DL, EVT VT, bool isTarget=false)
Create a ConstantFPSDNode wrapping a constant value.
LLVM_ABI bool haveNoCommonBitsSet(SDValue A, SDValue B) const
Return true if A and B have no common bits set.
LLVM_ABI SDValue getAddrSpaceCast(const SDLoc &dl, EVT VT, SDValue Ptr, unsigned SrcAS, unsigned DestAS, const SDNodeFlags Flags=SDNodeFlags())
Return an AddrSpaceCastSDNode.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI bool SignBitIsZeroFP(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero, for a floating-point value.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
SDValue getSetCC(const SDLoc &DL, EVT VT, SDValue LHS, SDValue RHS, ISD::CondCode Cond, SDValue Chain=SDValue(), bool IsSignaling=false, SDNodeFlags Flags={})
Helper function to make it easier to build SetCC's if you just have an ISD::CondCode instead of an SD...
LLVM_ABI SDValue getAtomic(unsigned Opcode, const SDLoc &dl, EVT MemVT, SDValue Chain, SDValue Ptr, SDValue Val, MachineMemOperand *MMO)
Gets a node for an atomic op, produces result (if relevant) and chain and takes 2 operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
std::pair< SDValue, SDValue > SplitVectorOperand(const SDNode *N, unsigned OpNo)
Split the node's operand with EXTRACT_SUBVECTOR and return the low/high part.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
LLVM_ABI std::pair< EVT, EVT > GetSplitDestVTs(const EVT &VT) const
Compute the VTs needed for the low/hi parts of a type which is split (or expanded) into two not neces...
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI SDValue getBitcastedAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by first bitcasting (from potentia...
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
SDValue getSelect(const SDLoc &DL, EVT VT, SDValue Cond, SDValue LHS, SDValue RHS, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build Select's if you just have operands and don't want to check...
LLVM_ABI void setNodeMemRefs(MachineSDNode *N, ArrayRef< MachineMemOperand * > NewMemRefs)
Mutate the specified machine node's memory references to the provided list.
LLVM_ABI SDValue getZeroExtendInReg(SDValue Op, const SDLoc &DL, EVT VT)
Return the expression required to zero extend the Op value assuming it was the smaller SrcTy value.
const DataLayout & getDataLayout() const
LLVM_ABI SDValue getTokenFactor(const SDLoc &DL, SmallVectorImpl< SDValue > &Vals)
Creates a new TokenFactor containing Vals.
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
LLVM_ABI SDValue getMemBasePlusOffset(SDValue Base, TypeSize Offset, const SDLoc &DL, const SDNodeFlags Flags=SDNodeFlags())
Returns sum of the base pointer and offset.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI void ReplaceAllUsesWith(SDValue From, SDValue To)
Modify anything using 'From' to use 'To' instead.
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI void RemoveDeadNode(SDNode *N)
Remove the specified node from the system.
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
SDValue getSelectCC(const SDLoc &DL, SDValue LHS, SDValue RHS, SDValue True, SDValue False, ISD::CondCode Cond, SDNodeFlags Flags=SDNodeFlags())
Helper function to make it easier to build SelectCC's if you just have an ISD::CondCode instead of an...
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getAnyExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either any-extending or truncat...
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI bool isKnownNeverNaN(SDValue Op, const APInt &DemandedElts, bool SNaN=false, unsigned Depth=0) const
Test whether the given SDValue (or all elements of it, if it is a vector) is known to never be NaN in...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
LLVM_ABI bool isBaseWithConstantOffset(SDValue Op) const
Return true if the specified operand is an ISD::ADD with a ConstantSDNode on the right-hand side,...
LLVM_ABI SDValue getVectorIdxConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getPOISON(EVT VT)
Return a POISON node. POISON does not have a useful SDLoc.
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getErrorMergeValues(ArrayRef< EVT > ResultTypes, SDValue Chain, const SDLoc &dl)
Return poison values for each of ResultTypes, substituting Chain for any result of type MVT::Other,...
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI SDValue getCondCode(ISD::CondCode Cond)
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
const SDValue & setRoot(SDValue N)
Set the current root tag of the SelectionDAG.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
int getMaskElt(unsigned Idx) const
ArrayRef< int > getMask() const
std::pair< iterator, bool > insert(PtrType Ptr)
Inserts Ptr if and only if there is no element in the container equal to Ptr.
SmallPtrSet - This class implements a set which is optimized for holding SmallSize or less elements.
size_type count(const T &V) const
count - Return 1 if the element is in the set, 0 otherwise.
Definition SmallSet.h:176
bool empty() const
Definition SmallSet.h:169
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void append(ItTy in_start, ItTy in_end)
Add the specified range to the end of the SmallVector.
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
Information about stack frame layout on the target.
Align getStackAlign() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
StackDirection getStackGrowthDirection() const
getStackGrowthDirection - Return the direction the stack grows
TargetInstrInfo - Interface to description of machine instruction set.
Type * Ty
Same as OrigTy, or partially legalized for soft float libcalls.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
virtual void finalizeLowering(MachineFunction &MF) const
Execute target specific actions to finalize target lowering.
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
virtual Align getPrefLoopAlignment(MachineLoop *ML=nullptr, const MachineBasicBlock *BlockToAlign=nullptr) const
Return the preferred loop alignment.
virtual unsigned getMaxPermittedBytesForAlignment(MachineBasicBlock *MBB) const
Return the maximum amount of bytes allowed to be emitted when padding for alignment.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
void setOperationPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
Convenience method to set an operation to Promote and specify the type in a single call.
LegalizeTypeAction
This enum indicates whether a types are legal for a target, and if not, what action should be used to...
void setHasExtractBitsInsn(bool hasExtractInsn=true)
Tells the code generator that the target has BitExtract instructions.
virtual TargetLoweringBase::LegalizeTypeAction getPreferredVectorAction(MVT VT) const
Return the preferred vector type legalization action.
virtual unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
Register getStackPointerRegisterToSaveRestore() const
If a physical register, this specifies the register that llvm.savestack/llvm.restorestack should save...
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
ExtractSubvectorCost
Enum that specifies how expensive lowering an EXTRACT_SUBVECTOR is.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
bool isOperationLegalOrCustom(unsigned Op, EVT VT, bool LegalOnly=false) const
Return true if the specified operation is legal on this target or can be made legal with custom lower...
virtual bool isNarrowingProfitable(SDNode *N, EVT SrcVT, EVT DestVT) const
Return true if it's profitable to narrow operations of type SrcVT to DestVT.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
void AddPromotedToType(unsigned Opc, MVT OrigVT, MVT DestVT)
If Opc/OrigVT is specified as being promoted, the promotion code defaults to trying a larger integer/...
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
bool allowsMemoryAccessForAlignment(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace=0, Align Alignment=Align(1), MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const
This function returns true if the memory access is aligned or if the target allows this specific unal...
virtual MVT getPointerMemTy(const DataLayout &DL, uint32_t AS=0) const
Return the in-memory pointer type for the given address space, defaults to the pointer type from the ...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
LegalizeAction getOperationAction(unsigned Op, EVT VT) const
Return how this operation should be treated: either it is legal, needs to be promoted to a larger siz...
SDValue scalarizeVectorStore(StoreSDNode *ST, SelectionDAG &DAG) const
std::vector< AsmOperandInfo > AsmOperandInfoVector
SDValue SimplifyMultipleUseDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, SelectionDAG &DAG, unsigned Depth=0) const
More limited version of SimplifyDemandedBits that can be used to "lookthrough" ops that don't contrib...
SDValue expandUnalignedStore(StoreSDNode *ST, SelectionDAG &DAG) const
Expands an unaligned store to 2 half-size stores for integer values, and possibly more for vectors.
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
bool parametersInCSRMatch(const MachineRegisterInfo &MRI, const uint32_t *CallerPreservedMask, const SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< SDValue > &OutVals) const
Check whether parameters to a call that are passed in callee saved registers are the same as from the...
std::pair< SDValue, SDValue > expandUnalignedLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Expands an unaligned load to 2 half-size loads for an integer, and possibly more for vectors.
SDValue expandFMINIMUMNUM_FMAXIMUMNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminimumnum/fmaximumnum into multiple comparison with selects.
virtual bool isTypeDesirableForOp(unsigned, EVT VT) const
Return true if the target has native support for the specified value type and it is 'desirable' to us...
virtual void computeKnownBitsForStackObjectPointer(KnownBits &Known, const MachineFunction &MF, Align Alignment) const
Determine known bits of a pointer to a known valid stack object.
std::pair< SDValue, SDValue > scalarizeVectorLoad(LoadSDNode *LD, SelectionDAG &DAG) const
Turn load of vector type into a load of the individual elements.
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
bool SimplifyDemandedBits(SDValue Op, const APInt &DemandedBits, const APInt &DemandedElts, KnownBits &Known, TargetLoweringOpt &TLO, unsigned Depth=0, bool AssumeSingleUse=false) const
Look at Op.
TargetLowering(const TargetLowering &)=delete
virtual MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *MBB) const
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
virtual AsmOperandInfoVector ParseConstraints(const DataLayout &DL, const TargetRegisterInfo *TRI, const CallBase &Call) const
Split up the constraint string from the inline assembly value into the specific constraints and their...
SDValue expandRoundInexactToOdd(EVT ResultVT, SDValue Op, const SDLoc &DL, SelectionDAG &DAG) const
Truncate Op to ResultVT.
virtual void ComputeConstraintToUse(AsmOperandInfo &OpInfo, SDValue Op, SelectionDAG *DAG=nullptr) const
Determines the constraint code and constraint type to use for the specific AsmOperandInfo,...
SDValue annotateStackObjectPointer(SDValue Ptr, SelectionDAG &DAG, const SDLoc &DL, Align Alignment) const
Annotate a stack object pointer with known-bits assertions.
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
SDValue expandFMINNUM_FMAXNUM(SDNode *N, SelectionDAG &DAG) const
Expand fminnum/fmaxnum into fminnum_ieee/fmaxnum_ieee with quieted inputs.
Primary interface to the complete machine description for the target machine.
CodeGenOptLevel getOptLevel() const
Returns the optimization level: None, Less, Default, or Aggressive.
const Triple & getTargetTriple() const
bool shouldAssumeDSOLocal(const GlobalValue *GV) const
TargetOptions Options
unsigned GuaranteedTailCallOpt
GuaranteedTailCallOpt - This flag is enabled when -tailcallopt is specified on the commandline.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Target - Wrapper for Target specific information.
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:48
OSType getOS() const
Get the parsed operating system type of this triple.
Definition Triple.h:524
Twine - A lightweight data structure for efficiently representing the concatenation of temporary valu...
Definition Twine.h:82
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:299
bool isBFloatTy() const
Return true if this is 'bfloat', a 16-bit bfloat type.
Definition Type.h:147
LLVM_ABI unsigned getPointerAddressSpace() const
Get the address space of this pointer or pointer vector type.
Type * getScalarType() const
If this is a vector type, return the element type, otherwise return 'this'.
Definition Type.h:363
bool isHalfTy() const
Return true if this is 'half', a 16-bit IEEE fp type.
Definition Type.h:144
bool isFunctionTy() const
True if this is an instance of FunctionType.
Definition Type.h:268
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
LLVM_ABI const fltSemantics & getFltSemantics() const
Definition Type.cpp:96
bool isVoidTy() const
Return true if this is 'void'.
Definition Type.h:141
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
LLVM_ABI unsigned getOperandNo() const
Return the operand # of this use in its User.
Definition Use.cpp:35
LLVM_ABI void set(Value *Val)
Definition Value.h:876
User * getUser() const
Returns the User that contains this Use.
Definition Use.h:61
const Use & getOperandUse(unsigned i) const
Definition User.h:220
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
LLVM_ABI void replaceAllUsesWith(Value *V)
Change all uses of this to point to a new Value.
Definition Value.cpp:553
LLVMContext & getContext() const
All values hold a context through their type.
Definition Value.h:260
iterator_range< user_iterator > users()
Definition Value.h:428
bool use_empty() const
Definition Value.h:348
iterator_range< use_iterator > uses()
Definition Value.h:382
Type * getElementType() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
constexpr bool isKnownEven() const
A return value of true indicates we know at compile time that the number of elements (vscale * Min) i...
Definition TypeSize.h:176
constexpr ScalarTy getKnownMinValue() const
Returns the minimum value this quantity can represent.
Definition TypeSize.h:165
self_iterator getIterator()
Definition ilist_node.h:123
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
@ CONSTANT_ADDRESS_32BIT
Address space for 32-bit constant memory.
@ BUFFER_STRIDED_POINTER
Address space for 192-bit fat buffer pointers with an additional index.
@ BARRIER
Address space for modeling barrier IDs as addresses.
@ REGION_ADDRESS
Address space for region memory. (GDS)
@ LOCAL_ADDRESS
Address space for local memory.
@ STREAMOUT_REGISTER
Internal address spaces. Can be freely renumbered.
@ CONSTANT_ADDRESS
Address space for constant memory (VTX2).
@ FLAT_ADDRESS
Address space for flat memory.
@ GLOBAL_ADDRESS
Address space for global memory (RAT0, VTX0).
@ BUFFER_FAT_POINTER
Address space for 160-bit buffer fat pointers.
@ PRIVATE_ADDRESS
Address space for private memory.
@ BUFFER_RESOURCE
Address space for 128-bit buffer resources.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr char NumVGPRs[]
Key for Kernel::CodeProps::Metadata::mNumVGPRs.
constexpr char Args[]
Key for Kernel::Metadata::mArgs.
constexpr char SymbolName[]
Key for Kernel::Metadata::mSymbolName.
bool isInlinableLiteralBF16(int16_t Literal, bool HasInv2Pi)
LLVM_READONLY const MIMGG16MappingInfo * getMIMGG16MappingInfo(unsigned G)
bool isInlinableLiteralFP16(int16_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isShader(CallingConv::ID CC)
bool shouldEmitConstantsToTextSection(const Triple &TT)
bool isFlatGlobalAddrSpace(unsigned AS)
const uint64_t FltRoundToHWConversionTable
bool isGFX12Plus(const MCSubtargetInfo &STI)
unsigned getNSAMaxSize(const MCSubtargetInfo &STI, bool HasSampler)
constexpr int64_t getNullPointerValue(unsigned AS)
Get the null pointer value for the given address space.
bool isGFX11(const MCSubtargetInfo &STI)
bool isGFX13(const MCSubtargetInfo &STI)
bool hasValueInRangeLikeMetadata(const MDNode &MD, int64_t Val)
Checks if Val is inside MD, a !range-like metadata.
LLVM_READNONE bool isLegalDPALU_DPPControl(const MCSubtargetInfo &ST, unsigned DC)
LLVM_READNONE constexpr bool mayTailCallThisCC(CallingConv::ID CC)
Return true if we might ever do TCO for calls with this calling convention.
unsigned getAMDHSACodeObjectVersion(const Module &M)
LLVM_READONLY bool hasNamedOperand(uint32_t Opcode, OpName NamedIdx)
int getMIMGOpcode(unsigned BaseOpcode, unsigned MIMGEncoding, unsigned VDataDwords, unsigned VAddrDwords, bool IndexedRsrc, bool IndexedSamp)
LLVM_READNONE constexpr bool isKernel(CallingConv::ID CC)
LLVM_READNONE constexpr bool isEntryFunctionCC(CallingConv::ID CC)
bool isInlinableLiteral32(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool isCompute(CallingConv::ID CC)
bool isIntrinsicSourceOfDivergence(unsigned IntrID)
LLVM_READNONE bool isInlinableIntLiteral(int64_t Literal)
Is this literal inlinable, and not one of the values intended for floating point values.
bool getMUBUFTfe(unsigned Opc)
TargetExtType * isNamedBarrier(const GlobalVariable &GV)
LLVM_READONLY int32_t getGlobalSaddrOp(uint32_t Opcode)
bool isGFX11Plus(const MCSubtargetInfo &STI)
std::optional< unsigned > getInlineEncodingV2F16(uint32_t Literal)
std::tuple< char, unsigned, unsigned > parseAsmConstraintPhysReg(StringRef Constraint)
Returns a valid charcode or 0 in the first entry if this is a valid physical register constraint.
bool isGFX10Plus(const MCSubtargetInfo &STI)
bool isValidWMMAScaleFmtCombination(unsigned AFmt, unsigned AScale, unsigned BFmt, unsigned BScale)
bool isUniformMMO(const MachineMemOperand *MMO)
std::optional< unsigned > getInlineEncodingV2I16(uint32_t Literal)
uint32_t decodeFltRoundToHWConversionTable(uint32_t FltRounds)
Read the hardware rounding mode equivalent of a AMDGPUFltRounds value.
bool isExtendedGlobalAddrSpace(unsigned AS)
LLVM_READONLY const MIMGDimInfo * getMIMGDimInfo(unsigned DimEnum)
std::optional< unsigned > getInlineEncodingV2BF16(uint32_t Literal)
LLVM_READONLY const MIMGBaseOpcodeInfo * getMIMGBaseOpcodeInfo(unsigned BaseOpcode)
unsigned getSyntheticApertureNumber(unsigned AS)
LLVM_READNONE constexpr bool isChainCC(CallingConv::ID CC)
int getMaskedMIMGOp(unsigned Opc, unsigned NewChannels)
const ImageDimIntrinsicInfo * getImageDimIntrinsicInfo(unsigned Intr)
bool isInlinableLiteralI16(int32_t Literal, bool HasInv2Pi)
LLVM_READNONE constexpr bool canGuaranteeTCO(CallingConv::ID CC)
LLVM_READNONE constexpr bool isGraphics(CallingConv::ID CC)
bool isInlinableLiteral64(int64_t Literal, bool HasInv2Pi)
Is this literal inlinable.
const RsrcIntrinsic * lookupRsrcIntrinsic(unsigned Intr)
const uint64_t FltRoundConversionTable
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ AMDGPU_CS
Used for Mesa/AMDPAL compute shaders.
@ AMDGPU_KERNEL
Used for AMDGPU code object kernels.
@ MaxID
The highest possible ID. Must be some 2^k - 1.
@ AMDGPU_Gfx
Used for AMD graphics targets.
@ AMDGPU_CS_ChainPreserve
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_CS_Chain
Used on AMDGPUs to give the middle-end more control over argument placement.
@ AMDGPU_PS
Used for Mesa/AMDPAL pixel shaders.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
Definition ISDOpcodes.h:263
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ PTRADD
PTRADD represents pointer arithmetic semantics, for targets that opt in using shouldPreservePtrArith(...
@ DELETED_NODE
DELETED_NODE - This is an illegal value that is used to catch errors.
Definition ISDOpcodes.h:47
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ SET_FPENV
Sets the current floating-point environment.
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ INSERT_SUBVECTOR
INSERT_SUBVECTOR(VECTOR1, VECTOR2, IDX) - Returns a vector with VECTOR2 inserted into VECTOR1.
Definition ISDOpcodes.h:605
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ FMAD
FMAD - Perform a * b + c, while getting the same result as the separately rounded operations.
Definition ISDOpcodes.h:527
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ ATOMIC_LOAD_USUB_COND
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ GlobalAddress
Definition ISDOpcodes.h:90
@ ATOMIC_CMP_SWAP_WITH_SUCCESS
Val, Success, OUTCHAIN = ATOMIC_CMP_SWAP_WITH_SUCCESS(INCHAIN, ptr, cmp, swap) N.b.
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:589
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ FP16_TO_FP
FP16_TO_FP, FP_TO_FP16 - These operators are used to perform promotions and truncation for half-preci...
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ FLDEXP
FLDEXP - ldexp, inspired by libm (op0 * 2**op1).
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ CONVERT_FROM_ARBITRARY_FP
CONVERT_FROM_ARBITRARY_FP - This operator converts from an arbitrary floating-point represented as an...
@ ATOMIC_LOAD_USUB_SAT
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ SET_ROUNDING
Set rounding mode.
Definition ISDOpcodes.h:993
@ CONVERGENCECTRL_GLUE
This does not correspond to any convergence control intrinsic.
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ READSTEADYCOUNTER
READSTEADYCOUNTER - This corresponds to the readfixedcounter intrinsic.
@ BR
Control flow instructions. These all have token chains.
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:355
@ FCANONICALIZE
Returns platform specific canonical encoding of a floating point number.
Definition ISDOpcodes.h:546
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:553
@ SSUBSAT
RESULT = [US]SUBSAT(LHS, RHS) - Perform saturation subtraction on 2 integers with the same bit width ...
Definition ISDOpcodes.h:377
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:249
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:232
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:351
@ CTLS
Count leading redundant sign bits.
Definition ISDOpcodes.h:810
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
Definition ISDOpcodes.h:988
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ GET_FPMODE
Reads the current dynamic floating-point control modes.
@ GET_FPENV
Gets the current floating-point environment.
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ EXTRACT_SUBVECTOR
EXTRACT_SUBVECTOR(VECTOR, IDX) - Returns a subvector from VECTOR.
Definition ISDOpcodes.h:619
@ FMINNUM_IEEE
FMINNUM_IEEE/FMAXNUM_IEEE - Perform floating-point minimumNumber or maximumNumber on two values,...
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:226
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ DEBUGTRAP
DEBUGTRAP - Trap intended to get the attention of a debugger.
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ ATOMIC_CMP_SWAP
Val, OUTCHAIN = ATOMIC_CMP_SWAP(INCHAIN, ptr, cmp, swap) For double-word atomic operations: ValLo,...
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ SMULO
Same for multiplication.
Definition ISDOpcodes.h:359
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ INLINEASM_BR
INLINEASM_BR - Branching version of inline asm. Used by asm-goto.
@ BF16_TO_FP
BF16_TO_FP, FP_TO_BF16 - These operators are used to perform promotions and truncation for bfloat16.
@ ATOMIC_LOAD_UDEC_WRAP
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
Definition ISDOpcodes.h:505
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ READCYCLECOUNTER
READCYCLECOUNTER - This corresponds to the readcyclecounter intrinsic.
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:510
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:805
@ ExternalSymbol
Definition ISDOpcodes.h:95
@ FFREXP
FFREXP - frexp, extract fractional and exponent component of a floating-point value.
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ SPONENTRY
SPONENTRY - Represents the llvm.sponentry intrinsic.
Definition ISDOpcodes.h:124
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ INLINEASM
INLINEASM - Represents an inline asm block.
@ FP_TO_SINT_SAT
FP_TO_[US]INT_SAT - Convert floating point value in operand 0 to a signed or unsigned scalar integer ...
Definition ISDOpcodes.h:963
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ BRCOND
BRCOND - Conditional branch.
@ CONVERT_TO_ARBITRARY_FP
CONVERT_TO_ARBITRARY_FP - Converts a native FP value to an arbitrary floating-point format,...
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ ATOMIC_LOAD_UINC_WRAP
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ SADDSAT
RESULT = [US]ADDSAT(LHS, RHS) - Perform saturation addition on 2 integers with the same bit width (W)...
Definition ISDOpcodes.h:368
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
bool isSignedIntSetCC(CondCode Code)
Return true if this is a setcc instruction that performs a signed comparison when used with integer o...
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
This namespace contains an enum with a value for every intrinsic/builtin function known by LLVM.
LLVM_ABI Function * getDeclarationIfExists(const Module *M, ID id)
Look up the Function declaration of the intrinsic id in the Module M and return it if it exists.
LLVM_ABI AttributeSet getFnAttributes(LLVMContext &C, ID id)
Return the function attributes for an intrinsic.
LLVM_ABI StringRef getBaseName(ID id)
Return the LLVM name for an intrinsic, without encoded types for overloading, such as "llvm....
LLVM_ABI AttributeList getAttributes(LLVMContext &C, ID id, FunctionType *FT)
Return the attributes for an intrinsic.
LLVM_ABI FunctionType * getType(LLVMContext &Context, ID id, ArrayRef< Type * > OverloadTys={})
Return the function type for an intrinsic.
BinaryOp_match< SpecificConstantMatch, SrcTy, TargetOpcode::G_SUB > m_Neg(const SrcTy &&Src)
Matches a register negated by a G_SUB.
bool mi_match(Reg R, const MachineRegisterInfo &MRI, Pattern &&P)
GFCstOrSplatGFCstMatch m_GFCstOrSplat(std::optional< FPValueAndVReg > &FPValReg)
BinaryOp_match< LHS, RHS, Instruction::Add > m_Add(const LHS &L, const RHS &R)
specificval_ty m_Specific(const Value *V)
Match if we have a specific specified value.
cst_pred_ty< is_one > m_One()
Match an integer 1 or a vector with all elements equal to 1.
specific_fpval m_SpecificFP(double V)
Match a specific floating point value or vector with all elements equal to the value.
auto m_Value()
Match an arbitrary value and ignore it.
auto m_FAbs(const Opnd0 &Op0)
BinaryOp_match< LHS, RHS, Instruction::Shl > m_Shl(const LHS &L, const RHS &R)
BinaryOp_match< LHS, RHS, Instruction::Sub > m_Sub(const LHS &L, const RHS &R)
auto m_IntrinsicWOChain(const OpndPreds &...Opnds)
bool sd_match(SDValue N, Pattern &&P)
ConstantInt_match m_ConstInt()
Match any integer constants or splat of an integer constant.
Offsets
Offsets in bytes from the start of the input buffer.
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
initializer< Ty > init(const Ty &Val)
constexpr double inv_pi
@ User
could "use" a pointer
DiagnosticInfoOptimizationBase::Argument NV
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
friend class Instruction
Iterator for Instructions in a `BasicBlock.
Definition BasicBlock.h:73
unsigned getOpcode(const VPValue *V)
Return the instruction opcode for the recipe defining V or 0 for unsupported recipes and VPValues not...
This is an optimization pass for GlobalISel generic memory operations.
GenericUniformityInfo< SSAContext > UniformityInfo
auto drop_begin(T &&RangeOrContainer, size_t N=1)
Return a range covering RangeOrContainer with the first N elements excluded.
Definition STLExtras.h:316
@ Offset
Definition DWP.cpp:577
LLVM_ABI void finalizeBundle(MachineBasicBlock &MBB, MachineBasicBlock::instr_iterator FirstMI, MachineBasicBlock::instr_iterator LastMI)
finalizeBundle - Finalize a machine instruction bundle which includes a sequence of instructions star...
bool all_of(R &&range, UnaryPredicate P)
Provide wrappers to std::all_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1755
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
detail::zippy< detail::zip_first, T, U, Args... > zip_equal(T &&t, U &&u, Args &&...args)
zip iterator that assumes that all iteratees have the same length.
Definition STLExtras.h:856
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
@ Implicit
Not emitted register (e.g. carry, or temporary result).
@ Dead
Unused definition.
@ Kill
The last use of a register.
@ Undef
Value of the register doesn't matter.
@ Define
Register definition.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Done
Definition Threading.h:60
bool CCAssignFn(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
CCAssignFn - This function assigns a location for Val, updating State to reflect the change.
@ Load
The value being inserted comes from a load (InsertElement only).
@ Store
The extracted value is stored (ExtractElement only).
constexpr int64_t minIntN(int64_t N)
Gets the minimum value for a N-bit signed integer.
Definition MathExtras.h:224
int bit_width(T Value)
Returns the number of bits needed to represent Value if Value is nonzero.
Definition bit.h:325
SDValue peekFPSignOps(SDValue Val)
Strip fabs/fneg/fcopysign from a value to get the underlying source.
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2224
constexpr T alignDown(U Value, V Align, W Skew=0)
Returns the largest unsigned integer less than or equal to Value and is Skew mod Align.
Definition MathExtras.h:541
MemoryEffectsBase< IRMemLocation > MemoryEffects
Summary of how a function affects memory in the program.
Definition ModRef.h:356
constexpr int popcount(T Value) noexcept
Count the number of set bits in a value.
Definition bit.h:156
LLVM_ABI ConstantFPSDNode * isConstOrConstSplatFP(SDValue N, bool AllowUndefs=false)
Returns the SDNode if it is a constant splat BuildVector or constant float.
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
uint64_t PowerOf2Ceil(uint64_t A)
Returns the power of two which is greater than or equal to the given value.
Definition MathExtras.h:380
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
constexpr bool isShiftedMask_64(uint64_t Value)
Return true if the argument contains a non-empty sequence of ones with the remainder zero (64 bit ver...
Definition MathExtras.h:274
constexpr T MinAlign(U A, V B)
A and B are either alignments or offsets.
Definition MathExtras.h:352
static const MachineMemOperand::Flags MONoClobber
Mark the MMO of a uniform load if there are no potentially clobbering stores on any path from the sta...
Definition SIInstrInfo.h:45
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1762
unsigned Log2_32(uint32_t Value)
Return the floor log base 2 of the specified value, -1 if the value is zero.
Definition MathExtras.h:326
AtomicOrderingCABI
Atomic ordering for C11 / C++11's memory models.
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
bool isBoolSGPR(SDValue V)
MachineInstr * getImm(const MachineOperand &MO, const MachineRegisterInfo *MRI)
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
constexpr uint32_t Hi_32(uint64_t Value)
Return the high 32 bits of a 64 bit value.
Definition MathExtras.h:151
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
constexpr uint32_t Lo_32(uint64_t Value)
Return the low 32 bits of a 64 bit value.
Definition MathExtras.h:156
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
static const MachineMemOperand::Flags MOCooperative
Mark the MMO of cooperative load/store atomics.
Definition SIInstrInfo.h:53
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
LLVM_ABI Value * buildAtomicRMWValue(AtomicRMWInst::BinOp Op, IRBuilderBase &Builder, Value *Loaded, Value *Val)
Emit IR to implement the given atomicrmw operation on values in registers, returning the new value.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ AfterLegalizeDAG
Definition DAGCombine.h:19
@ AfterLegalizeVectorOps
Definition DAGCombine.h:18
@ AfterLegalizeTypes
Definition DAGCombine.h:17
@ Or
Bitwise or logical OR of integers.
@ Mul
Product of integers.
@ Add
Sum of integers.
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
DWARFExpression::Operation Op
RoundingMode
Rounding mode.
@ NearestTiesToEven
roundTiesToEven.
unsigned M0(unsigned Val)
Definition VE.h:376
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr int64_t maxIntN(int64_t N)
Gets the maximum value for a N-bit signed integer.
Definition MathExtras.h:233
constexpr unsigned BitWidth
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
LLVM_ABI std::pair< Value *, Value * > buildCmpXchgValue(IRBuilderBase &Builder, Value *Ptr, Value *Cmp, Value *Val, Align Alignment, bool IsVolatile=false)
Emit IR to implement the given cmpxchg operation on values in registers, returning the new value.
LLVM_ABI std::optional< ValueAndVReg > getIConstantVRegValWithLookThrough(Register VReg, const MachineRegisterInfo &MRI, bool LookThroughInstrs=true)
If VReg is defined by a statically evaluable chain of instructions rooted on a G_CONSTANT returns its...
Definition Utils.cpp:436
auto find_if(R &&Range, UnaryPredicate P)
Provide wrappers to std::find_if which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1788
std::optional< StringRef > getAtomicScopeIRString(const Triple &T, AtomicScope S, bool IsSingleAddressSpace=false)
Returns the LLVM IR syncscope string that T uses to spell S.
Definition AtomicScope.h:34
LLVM_ABI bool isOneConstant(SDValue V)
Returns true if V is a constant integer one.
static const MachineMemOperand::Flags MOLastUse
Mark the MMO of a load as the last use.
Definition SIInstrInfo.h:49
bool is_contained(R &&Range, const E &Element)
Returns true if Element is found in Range.
Definition STLExtras.h:1963
Align commonAlignment(Align A, uint64_t Offset)
Returns the alignment that satisfies both alignments.
Definition Alignment.h:201
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Next
Definition InstrProf.h:147
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
constexpr RegState getUndefRegState(bool B)
@ Custom
The result value requires a custom uniformity check.
Definition Uniformity.h:31
LLVM_ABI Printable printReg(Register Reg, const TargetRegisterInfo *TRI=nullptr, unsigned SubIdx=0, const MachineRegisterInfo *MRI=nullptr)
Prints virtual and physical registers with or without a TRI instance.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
int64_t DWordOffset
int64_t PermMask
std::tuple< const ArgDescriptor *, const TargetRegisterClass *, LLT > getPreloadedValue(PreloadedValue Value) const
static const AMDGPUFunctionArgInfo FixedABIFunctionInfo
static constexpr uint64_t encode(Fields... Values)
static std::tuple< typename Fields::ValueType... > decode(uint64_t Encoded)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
MCRegister getRegister() const
static ArgDescriptor createArg(const ArgDescriptor &Arg, unsigned Mask)
static ArgDescriptor createRegister(Register Reg, unsigned Mask=~0u)
Helper struct shared between Function Specialization and SCCP Solver.
Definition SCCPSolver.h:42
Represents the full denormal controls for a function, including the default mode and the f32 specific...
Represent subnormal handling kind for floating point instruction inputs and outputs.
@ Dynamic
Denormals have unknown treatment.
static constexpr DenormalMode getPreserveSign()
static constexpr DenormalMode getIEEE()
Extended Value Type.
Definition ValueTypes.h:35
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
EVT changeTypeToInteger() const
Return the type converted to an equivalently sized integer or vector with integer element type.
Definition ValueTypes.h:129
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
bool isByteSized() const
Return true if the bit size is a multiple of 8.
Definition ValueTypes.h:266
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
bool isPow2VectorType() const
Returns true if the given vector is a power of 2.
Definition ValueTypes.h:501
TypeSize getStoreSizeInBits() const
Return the number of bits overwritten by a store of the specified value type.
Definition ValueTypes.h:435
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
EVT changeElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a type whose attributes match ourselves with the exception of the element type that i...
Definition ValueTypes.h:121
bool isVectorOf(EVT EltVT) const
Return true if this is a vector with matching element type.
Definition ValueTypes.h:181
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
LLVM_ABI const fltSemantics & getFltSemantics() const
Returns an APFloat semantics tag appropriate for the value type.
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
unsigned getPointerAddrSpace() const
unsigned getByValSize() const
Align getNonZeroMemAlign() const
InputArg - This struct carries flags and type information about a single incoming (formal) argument o...
MVT VT
Legalized type of this argument part.
unsigned getOrigArgIndex() const
OutputArg - This struct carries flags and a value for a single outgoing (actual) argument or outgoing...
static LLVM_ABI std::optional< bool > eq(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_EQ result.
bool isUnknown() const
Returns true if we don't know any bits.
Definition KnownBits.h:64
KnownBits trunc(unsigned BitWidth) const
Return known bits for a truncation of the value we're tracking.
Definition KnownBits.h:165
static KnownBits add(const KnownBits &LHS, const KnownBits &RHS, bool NSW=false, bool NUW=false, bool SelfAdd=false)
Compute knownbits resulting from addition of LHS and RHS.
Definition KnownBits.h:361
unsigned countMinLeadingZeros() const
Returns the minimum number of leading zero bits.
Definition KnownBits.h:262
static LLVM_ABI std::optional< bool > ule(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_ULE result.
static LLVM_ABI std::optional< bool > uge(const KnownBits &LHS, const KnownBits &RHS)
Determine if these known bits always give the same ICMP_UGE result.
bool isKnownNeverNaN() const
Return true if it's known this can never be a nan.
static LLVM_ABI KnownFPClass bitcast(const fltSemantics &FltSemantics, const KnownBits &Bits)
Report known values for a bitcast into a float with provided semantics.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This struct is a compact representation of a valid (power of two) or undefined (0) alignment.
Definition Alignment.h:106
These are IR-level optimization flags that may be propagated to SDNodes.
bool hasNoUnsignedWrap() const
bool hasAllowContract() const
bool hasNoSignedWrap() const
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
unsigned int NumVTs
DenormalMode FP64FP16Denormals
If this is set, neither input or output denormals are flushed for both f64 and f16/v2f16 instructions...
DenormalMode FP32Denormals
If this is set, neither input or output denormals are flushed for most f32 instructions.
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
std::optional< unsigned > fallbackAddressSpace
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
SmallVector< ISD::OutputArg, 32 > Outs