LLVM 24.0.0git
X86ISelLoweringCall.cpp
Go to the documentation of this file.
1//===- llvm/lib/Target/X86/X86ISelCallLowering.cpp - Call lowering --------===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9/// \file
10/// This file implements the lowering of LLVM calls to DAG nodes.
11//
12//===----------------------------------------------------------------------===//
13
15#include "X86.h"
16#include "X86CallingConv.h"
17#include "X86FrameLowering.h"
18#include "X86ISelLowering.h"
19#include "X86InstrBuilder.h"
21#include "X86TargetMachine.h"
22#include "llvm/ADT/Statistic.h"
28#include "llvm/IR/IRBuilder.h"
29#include "llvm/IR/Module.h"
31
32#define DEBUG_TYPE "x86-isel"
33
34using namespace llvm;
35
36STATISTIC(NumTailCalls, "Number of tail calls");
37
38/// Call this when the user attempts to do something unsupported, like
39/// returning a double without SSE2 enabled on x86_64. This is not fatal, unlike
40/// report_fatal_error, so calling code should attempt to recover without
41/// crashing.
42static void errorUnsupported(SelectionDAG &DAG, const SDLoc &dl,
43 const char *Msg) {
45 DAG.getContext()->diagnose(
47}
48
49/// Returns true if a CC can dynamically exclude a register from the list of
50/// callee-saved-registers (TargetRegistryInfo::getCalleeSavedRegs()) based on
51/// the return registers.
53 switch (CC) {
54 default:
55 return false;
59 return true;
60 }
61}
62
63/// Returns true if a CC can dynamically exclude a register from the list of
64/// callee-saved-registers (TargetRegistryInfo::getCalleeSavedRegs()) based on
65/// the parameters.
69
70static std::pair<MVT, unsigned>
72 const X86Subtarget &Subtarget) {
73 // v2i1/v4i1/v8i1/v16i1 all pass in xmm registers unless the calling
74 // convention is one that uses k registers.
75 if (NumElts == 2)
76 return {MVT::v2i64, 1};
77 if (NumElts == 4)
78 return {MVT::v4i32, 1};
79 if (NumElts == 8 && CC != CallingConv::X86_RegCall &&
81 return {MVT::v8i16, 1};
82 if (NumElts == 16 && CC != CallingConv::X86_RegCall &&
84 return {MVT::v16i8, 1};
85 // v32i1 passes in ymm unless we have BWI and the calling convention is
86 // regcall.
87 if (NumElts == 32 && (!Subtarget.hasBWI() || CC != CallingConv::X86_RegCall))
88 return {MVT::v32i8, 1};
89 // Split v64i1 vectors if we don't have v64i8 available.
90 if (NumElts == 64 && Subtarget.hasBWI() && CC != CallingConv::X86_RegCall) {
91 if (Subtarget.useAVX512Regs())
92 return {MVT::v64i8, 1};
93 return {MVT::v32i8, 2};
94 }
95
96 // Break wide or odd vXi1 vectors into scalars to match avx2 behavior.
97 if (!isPowerOf2_32(NumElts) || (NumElts == 64 && !Subtarget.hasBWI()) ||
98 NumElts > 64)
99 return {MVT::i8, NumElts};
100
102}
103
106 EVT VT) const {
107 if (VT.isVector()) {
108 if (VT.getVectorElementType() == MVT::i1 && Subtarget.hasAVX512()) {
109 unsigned NumElts = VT.getVectorNumElements();
110
111 MVT RegisterVT;
112 unsigned NumRegisters;
113 std::tie(RegisterVT, NumRegisters) =
114 handleMaskRegisterForCallingConv(NumElts, CC, Subtarget);
115 if (RegisterVT != MVT::INVALID_SIMPLE_VALUE_TYPE)
116 return RegisterVT;
117 }
118
119 if (VT.getVectorElementType() == MVT::f16 && VT.getVectorNumElements() < 8)
120 return MVT::v8f16;
121 }
122
123 // We will use more GPRs for f64 and f80 on 32 bits when x87 is disabled.
124 if ((VT == MVT::f64 || VT == MVT::f80) && !Subtarget.is64Bit() &&
125 !Subtarget.hasX87())
126 return MVT::i32;
127
128 if (isTypeLegal(MVT::f16)) {
129 if (VT.isVectorOf(MVT::bf16))
131 Context, CC, VT.changeVectorElementType(Context, MVT::f16));
132
133 if (VT == MVT::bf16)
134 return MVT::f16;
135 }
136
137 return TargetLowering::getRegisterTypeForCallingConv(Context, CC, VT);
138}
139
142 EVT VT) const {
143 if (VT.isVector()) {
144 if (VT.getVectorElementType() == MVT::i1 && Subtarget.hasAVX512()) {
145 unsigned NumElts = VT.getVectorNumElements();
146
147 MVT RegisterVT;
148 unsigned NumRegisters;
149 std::tie(RegisterVT, NumRegisters) =
150 handleMaskRegisterForCallingConv(NumElts, CC, Subtarget);
151 if (RegisterVT != MVT::INVALID_SIMPLE_VALUE_TYPE)
152 return NumRegisters;
153 }
154
155 if (VT.getVectorElementType() == MVT::f16 && VT.getVectorNumElements() < 8)
156 return 1;
157 }
158
159 // We have to split f64 to 2 registers and f80 to 3 registers on 32 bits if
160 // x87 is disabled.
161 if (!Subtarget.is64Bit() && !Subtarget.hasX87()) {
162 if (VT == MVT::f64)
163 return 2;
164 if (VT == MVT::f80)
165 return 3;
166 }
167
168 if (VT.isVectorOf(MVT::bf16) && isTypeLegal(MVT::f16))
170 Context, CC, VT.changeVectorElementType(Context, MVT::f16));
171
172 return TargetLowering::getNumRegistersForCallingConv(Context, CC, VT);
173}
174
176 LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT,
177 unsigned &NumIntermediates, MVT &RegisterVT) const {
178 // Break wide or odd vXi1 vectors into scalars to match avx2 behavior.
179 if (VT.isVectorOf(MVT::i1) && Subtarget.hasAVX512() &&
181 (VT.getVectorNumElements() == 64 && !Subtarget.hasBWI()) ||
182 VT.getVectorNumElements() > 64)) {
183 RegisterVT = MVT::i8;
184 IntermediateVT = MVT::i1;
185 NumIntermediates = VT.getVectorNumElements();
186 return NumIntermediates;
187 }
188
189 // Split v64i1 vectors if we don't have v64i8 available.
190 if (VT == MVT::v64i1 && Subtarget.hasBWI() && !Subtarget.useAVX512Regs() &&
192 RegisterVT = MVT::v32i8;
193 IntermediateVT = MVT::v32i1;
194 NumIntermediates = 2;
195 return 2;
196 }
197
198 // Split vNbf16 vectors according to vNf16.
199 if (VT.isVectorOf(MVT::bf16) && isTypeLegal(MVT::f16))
200 VT = VT.changeVectorElementType(Context, MVT::f16);
201
202 return TargetLowering::getVectorTypeBreakdownForCallingConv(Context, CC, VT, IntermediateVT,
203 NumIntermediates, RegisterVT);
204}
205
207 LLVMContext& Context,
208 EVT VT) const {
209 if (!VT.isVector())
210 return MVT::i8;
211
212 if (Subtarget.hasAVX512()) {
213 // Figure out what this type will be legalized to.
214 EVT LegalVT = VT;
215 while (getTypeAction(Context, LegalVT) != TypeLegal)
216 LegalVT = getTypeToTransformTo(Context, LegalVT);
217
218 // If we got a 512-bit vector then we'll definitely have a vXi1 compare.
219 if (LegalVT.getSimpleVT().is512BitVector())
220 return EVT::getVectorVT(Context, MVT::i1, VT.getVectorElementCount());
221
222 if (LegalVT.getSimpleVT().isVector() && Subtarget.hasVLX()) {
223 // If we legalized to less than a 512-bit vector, then we will use a vXi1
224 // compare for vXi32/vXi64 for sure. If we have BWI we will also support
225 // vXi16/vXi8.
226 MVT EltVT = LegalVT.getSimpleVT().getVectorElementType();
227 if (Subtarget.hasBWI() || EltVT.getSizeInBits() >= 32)
228 return EVT::getVectorVT(Context, MVT::i1, VT.getVectorElementCount());
229 }
230 }
231
233}
234
236 Type *Ty, CallingConv::ID CallConv, bool isVarArg,
237 const DataLayout &DL) const {
238 // On x86-64 i128 is split into two i64s and needs to be allocated to two
239 // consecutive registers, or spilled to the stack as a whole. On x86-32 i128
240 // is split to four i32s and never actually passed in registers, but we use
241 // the consecutive register mark to match it in TableGen.
242 if (Ty->isIntegerTy(128))
243 return true;
244
245 // On x86-32, fp128 acts the same as i128.
246 if (Subtarget.is32Bit() && Ty->isFP128Ty())
247 return true;
248
249 return false;
250}
251
252/// Helper for getByValTypeAlignment to determine
253/// the desired ByVal argument alignment.
254static void getMaxByValAlign(Type *Ty, Align &MaxAlign) {
255 if (MaxAlign == 16)
256 return;
257 if (VectorType *VTy = dyn_cast<VectorType>(Ty)) {
258 if (VTy->getPrimitiveSizeInBits().getFixedValue() == 128)
259 MaxAlign = Align(16);
260 } else if (ArrayType *ATy = dyn_cast<ArrayType>(Ty)) {
261 Align EltAlign;
262 getMaxByValAlign(ATy->getElementType(), EltAlign);
263 if (EltAlign > MaxAlign)
264 MaxAlign = EltAlign;
265 } else if (StructType *STy = dyn_cast<StructType>(Ty)) {
266 for (auto *EltTy : STy->elements()) {
267 Align EltAlign;
268 getMaxByValAlign(EltTy, EltAlign);
269 if (EltAlign > MaxAlign)
270 MaxAlign = EltAlign;
271 if (MaxAlign == 16)
272 break;
273 }
274 }
275}
276
277/// Return the desired alignment for ByVal aggregate
278/// function arguments in the caller parameter area. For X86, aggregates
279/// that contain SSE vectors are placed at 16-byte boundaries while the rest
280/// are at 4-byte boundaries.
282 const DataLayout &DL) const {
283 if (Subtarget.is64Bit())
284 return std::max(DL.getABITypeAlign(Ty), Align::Constant<8>());
285
286 Align Alignment(4);
287 if (Subtarget.hasSSE1())
288 getMaxByValAlign(Ty, Alignment);
289 return Alignment;
290}
291
292/// It returns EVT::Other if the type should be determined using generic
293/// target-independent logic.
294/// For vector ops we check that the overall size isn't larger than our
295/// preferred vector width.
297 LLVMContext &Context, const MemOp &Op,
298 const AttributeList &FuncAttributes) const {
299 if (!FuncAttributes.hasFnAttr(Attribute::NoImplicitFloat)) {
300 if (Op.size() >= 16 &&
301 (!Subtarget.isUnalignedMem16Slow() || Op.isAligned(Align(16)))) {
302 // FIXME: Check if unaligned 64-byte accesses are slow.
303 if (Op.size() >= 64 && Subtarget.hasAVX512() &&
304 (Subtarget.getPreferVectorWidth() >= 512)) {
305 return Subtarget.hasBWI() ? MVT::v64i8 : MVT::v16i32;
306 }
307 // FIXME: Check if unaligned 32-byte accesses are slow.
308 if (Op.size() >= 32 && Subtarget.hasAVX() &&
309 Subtarget.useLight256BitInstructions()) {
310 // Although this isn't a well-supported type for AVX1, we'll let
311 // legalization and shuffle lowering produce the optimal codegen. If we
312 // choose an optimal type with a vector element larger than a byte,
313 // getMemsetStores() may create an intermediate splat (using an integer
314 // multiply) before we splat as a vector.
315 return MVT::v32i8;
316 }
317 if (Subtarget.hasSSE2() && (Subtarget.getPreferVectorWidth() >= 128))
318 return MVT::v16i8;
319 // TODO: Can SSE1 handle a byte vector?
320 // If we have SSE1 registers we should be able to use them.
321 if (Subtarget.hasSSE1() && (Subtarget.is64Bit() || Subtarget.hasX87()) &&
322 (Subtarget.getPreferVectorWidth() >= 128))
323 return MVT::v4f32;
324 } else if (((Op.isMemcpyOrMemmove() && !Op.isMemcpyStrSrc()) ||
325 Op.isZeroMemset()) &&
326 Op.size() >= 8 && !Subtarget.is64Bit() && Subtarget.hasSSE2()) {
327 // Do not use f64 to lower memcpy if source is string constant. It's
328 // better to use i32 to avoid the loads.
329 // Also, do not use f64 to lower memset unless this is a memset of zeros.
330 // The gymnastics of splatting a byte value into an XMM register and then
331 // only using 8-byte stores (because this is a CPU with slow unaligned
332 // 16-byte accesses) makes that a loser.
333 return MVT::f64;
334 }
335 }
336 // This is a compromise. If we reach here, unaligned accesses may be slow on
337 // this target. However, creating smaller, aligned accesses could be even
338 // slower and would certainly be a lot more code.
339 if (Subtarget.is64Bit() && Op.size() >= 8)
340 return MVT::i64;
341 return MVT::i32;
342}
343
345 if (VT == MVT::f32)
346 return Subtarget.hasSSE1();
347 if (VT == MVT::f64)
348 return Subtarget.hasSSE2();
349 return true;
350}
351
352static bool isBitAligned(Align Alignment, uint64_t SizeInBits) {
353 return (8 * Alignment.value()) % SizeInBits == 0;
354}
355
357 if (isBitAligned(Alignment, VT.getSizeInBits()))
358 return true;
359 switch (VT.getSizeInBits()) {
360 default:
361 // 8-byte and under are always assumed to be fast.
362 return true;
363 case 128:
364 return !Subtarget.isUnalignedMem16Slow();
365 case 256:
366 return !Subtarget.isUnalignedMem32Slow();
367 // TODO: What about AVX-512 (512-bit) accesses?
368 }
369}
370
372 EVT VT, unsigned, Align Alignment, MachineMemOperand::Flags Flags,
373 unsigned *Fast) const {
374 if (Fast)
375 *Fast = isMemoryAccessFast(VT, Alignment);
376 // NonTemporal vector memory ops must be aligned.
377 if (!!(Flags & MachineMemOperand::MONonTemporal) && VT.isVector()) {
378 // NT loads can only be vector aligned, so if its less aligned than the
379 // minimum vector size (which we can split the vector down to), we might as
380 // well use a regular unaligned vector load.
381 // We don't have any NT loads pre-SSE41.
382 if (!!(Flags & MachineMemOperand::MOLoad))
383 return (Alignment < 16 || !Subtarget.hasSSE41());
384 return false;
385 }
386 // Misaligned accesses of any size are always allowed.
387 return true;
388}
389
391 const DataLayout &DL, EVT VT,
392 unsigned AddrSpace, Align Alignment,
394 unsigned *Fast) const {
395 if (Fast)
396 *Fast = isMemoryAccessFast(VT, Alignment);
397 if (!!(Flags & MachineMemOperand::MONonTemporal) && VT.isVector()) {
398 if (allowsMisalignedMemoryAccesses(VT, AddrSpace, Alignment, Flags,
399 /*Fast=*/nullptr))
400 return true;
401 // NonTemporal vector memory ops are special, and must be aligned.
402 if (!isBitAligned(Alignment, VT.getSizeInBits()))
403 return false;
404 switch (VT.getSizeInBits()) {
405 case 128:
406 if (!!(Flags & MachineMemOperand::MOLoad) && Subtarget.hasSSE41())
407 return true;
408 if (!!(Flags & MachineMemOperand::MOStore) && Subtarget.hasSSE2())
409 return true;
410 return false;
411 case 256:
412 if (!!(Flags & MachineMemOperand::MOLoad) && Subtarget.hasAVX2())
413 return true;
414 if (!!(Flags & MachineMemOperand::MOStore) && Subtarget.hasAVX())
415 return true;
416 return false;
417 case 512:
418 if (Subtarget.hasAVX512())
419 return true;
420 return false;
421 default:
422 return false; // Don't have NonTemporal vector memory ops of this size.
423 }
424 }
425 return true;
426}
427
428/// Return the entry encoding for a jump table in the
429/// current function. The returned value is a member of the
430/// MachineJumpTableInfo::JTEntryKind enum.
432 // In GOT pic mode, each entry in the jump table is emitted as a @GOTOFF
433 // symbol.
434 if (isPositionIndependent() && Subtarget.isPICStyleGOT())
436 if (isPositionIndependent() &&
438 !Subtarget.isTargetCOFF())
440
441 // Otherwise, use the normal jump table encoding heuristics.
443}
444
446 return Subtarget.useSoftFloat();
447}
448
450 ArgListTy &Args) const {
451
452 // Only relabel X86-32 for C / Stdcall CCs.
453 if (Subtarget.is64Bit())
454 return;
455 if (CC != CallingConv::C && CC != CallingConv::X86_StdCall)
456 return;
457 unsigned ParamRegs = 0;
458 if (auto *M = MF->getFunction().getParent())
459 ParamRegs = M->getNumberRegisterParameters();
460
461 // Mark the first N int arguments as having reg
462 for (auto &Arg : Args) {
463 Type *T = Arg.Ty;
464 if (T->isIntOrPtrTy())
465 if (MF->getDataLayout().getTypeAllocSize(T) <= 8) {
466 unsigned numRegs = 1;
467 if (MF->getDataLayout().getTypeAllocSize(T) > 4)
468 numRegs = 2;
469 if (ParamRegs < numRegs)
470 return;
471 ParamRegs -= numRegs;
472 Arg.IsInReg = true;
473 }
474 }
475}
476
477const MCExpr *
479 const MachineBasicBlock *MBB,
480 unsigned uid,MCContext &Ctx) const{
481 assert(isPositionIndependent() && Subtarget.isPICStyleGOT());
482 // In 32-bit ELF systems, our jump table entries are formed with @GOTOFF
483 // entries.
484 return MCSymbolRefExpr::create(MBB->getSymbol(), X86::S_GOTOFF, Ctx);
485}
486
487/// Returns relocation base for the given PIC jumptable.
489 SelectionDAG &DAG) const {
490 if (!Subtarget.is64Bit())
491 // This doesn't have SDLoc associated with it, but is not really the
492 // same as a Register.
493 return DAG.getNode(X86ISD::GlobalBaseReg, SDLoc(),
495 return Table;
496}
497
498/// This returns the relocation base for the given PIC jumptable,
499/// the same as getPICJumpTableRelocBase, but as an MCExpr.
501getPICJumpTableRelocBaseExpr(const MachineFunction *MF, unsigned JTI,
502 MCContext &Ctx) const {
503 // X86-64 uses RIP relative addressing based on the jump table label.
504 if (Subtarget.isPICStyleRIPRel() ||
505 (Subtarget.is64Bit() &&
508
509 // Otherwise, the reference is relative to the PIC base.
510 return MCSymbolRefExpr::create(MF->getPICBaseSymbol(), Ctx);
511}
512
513std::pair<const TargetRegisterClass *, uint8_t>
515 MVT VT) const {
516 const TargetRegisterClass *RRC = nullptr;
517 uint8_t Cost = 1;
518 switch (VT.SimpleTy) {
519 default:
521 case MVT::i8: case MVT::i16: case MVT::i32: case MVT::i64:
522 RRC = Subtarget.is64Bit() ? &X86::GR64RegClass : &X86::GR32RegClass;
523 break;
524 case MVT::x86mmx:
525 RRC = &X86::VR64RegClass;
526 break;
527 case MVT::f32: case MVT::f64:
528 case MVT::v16i8: case MVT::v8i16: case MVT::v4i32: case MVT::v2i64:
529 case MVT::v4f32: case MVT::v2f64:
530 case MVT::v32i8: case MVT::v16i16: case MVT::v8i32: case MVT::v4i64:
531 case MVT::v8f32: case MVT::v4f64:
532 case MVT::v64i8: case MVT::v32i16: case MVT::v16i32: case MVT::v8i64:
533 case MVT::v16f32: case MVT::v8f64:
534 RRC = &X86::VR128XRegClass;
535 break;
536 }
537 return std::make_pair(RRC, Cost);
538}
539
540unsigned X86TargetLowering::getAddressSpace() const {
541 if (Subtarget.is64Bit())
543 : X86AS::FS;
544 return X86AS::GS;
545}
546
547static bool hasStackGuardSlotTLS(const Triple &TargetTriple) {
548 return TargetTriple.isOSGlibc() || TargetTriple.isMusl() ||
549 TargetTriple.isOSFuchsia() || TargetTriple.isAndroid();
550}
551
558
559Value *
561 const LibcallLoweringInfo &Libcalls) const {
562 // glibc, bionic, and Fuchsia have a special slot for the stack guard in
563 // tcbhead_t; use it instead of the usual global variable (see
564 // sysdeps/{i386,x86_64}/nptl/tls.h)
565 if (hasStackGuardSlotTLS(Subtarget.getTargetTriple())) {
566 unsigned AddressSpace = getAddressSpace();
567
568 // <zircon/tls.h> defines ZX_TLS_STACK_GUARD_OFFSET with this value.
569 if (Subtarget.isTargetFuchsia())
570 return SegmentOffset(IRB, 0x10, AddressSpace);
571
572 Module *M = IRB.GetInsertBlock()->getParent()->getParent();
573 // Specially, some users may customize the base reg and offset.
574 int Offset = M->getStackProtectorGuardOffset();
575 // If we don't set -stack-protector-guard-offset value:
576 // %fs:0x28, unless we're using a Kernel code model, in which case
577 // it's %gs:0x28. gs:0x14 on i386.
578 if (Offset == INT_MAX)
579 Offset = (Subtarget.is64Bit()) ? 0x28 : 0x14;
580
581 StringRef GuardReg = M->getStackProtectorGuardReg();
582 if (GuardReg == "fs")
584 else if (GuardReg == "gs")
586
587 // Use symbol guard if user specify.
588 StringRef GuardSymb = M->getStackProtectorGuardSymbol();
589 if (!GuardSymb.empty()) {
590 GlobalVariable *GV = M->getGlobalVariable(GuardSymb);
591 if (!GV) {
592 Type *Ty = Subtarget.is64Bit() ? Type::getInt64Ty(M->getContext())
593 : Type::getInt32Ty(M->getContext());
594 GV = new GlobalVariable(*M, Ty, false, GlobalValue::ExternalLinkage,
595 nullptr, GuardSymb, nullptr,
597 if (!Subtarget.isTargetDarwin())
598 GV->setDSOLocal(M->getDirectAccessExternalData());
599 }
600 return GV;
601 }
602
603 return SegmentOffset(IRB, Offset, AddressSpace);
604 }
605 return TargetLowering::getIRStackGuard(IRB, Libcalls);
606}
607
609 Module &M, const LibcallLoweringInfo &Libcalls) const {
610 // MSVC CRT provides functionalities for stack protection.
611 RTLIB::LibcallImpl SecurityCheckCookieLibcall =
612 Libcalls.getLibcallImpl(RTLIB::SECURITY_CHECK_COOKIE);
613
614 RTLIB::LibcallImpl SecurityCookieVar =
615 Libcalls.getLibcallImpl(RTLIB::STACK_CHECK_GUARD);
616 if (SecurityCheckCookieLibcall != RTLIB::Unsupported &&
617 SecurityCookieVar != RTLIB::Unsupported) {
618 // MSVC CRT provides functionalities for stack protection.
619 // MSVC CRT has a global variable holding security cookie.
620 M.getOrInsertGlobal(getLibcallImplName(SecurityCookieVar),
621 PointerType::getUnqual(M.getContext()));
622
623 // MSVC CRT has a function to validate security cookie.
624 FunctionCallee SecurityCheckCookie =
625 M.getOrInsertFunction(getLibcallImplName(SecurityCheckCookieLibcall),
626 Type::getVoidTy(M.getContext()),
627 PointerType::getUnqual(M.getContext()));
628
629 if (Function *F = dyn_cast<Function>(SecurityCheckCookie.getCallee())) {
630 F->setCallingConv(CallingConv::X86_FastCall);
631 F->addParamAttr(0, Attribute::AttrKind::InReg);
632 }
633 return;
634 }
635
636 StringRef GuardMode = M.getStackProtectorGuard();
637
638 // glibc, bionic, and Fuchsia have a special slot for the stack guard.
639 if ((GuardMode == "tls" || GuardMode.empty()) &&
640 hasStackGuardSlotTLS(Subtarget.getTargetTriple()))
641 return;
643}
644
646 IRBuilderBase &IRB, const LibcallLoweringInfo &Libcalls) const {
647 // Android provides a fixed TLS slot for the SafeStack pointer. See the
648 // definition of TLS_SLOT_SAFESTACK in
649 // https://android.googlesource.com/platform/bionic/+/master/libc/private/bionic_tls.h
650 if (Subtarget.isTargetAndroid()) {
651 // %fs:0x48, unless we're using a Kernel code model, in which case it's %gs:
652 // %gs:0x24 on i386
653 int Offset = (Subtarget.is64Bit()) ? 0x48 : 0x24;
654 return SegmentOffset(IRB, Offset, getAddressSpace());
655 }
656
657 // Fuchsia is similar.
658 if (Subtarget.isTargetFuchsia()) {
659 // <zircon/tls.h> defines ZX_TLS_UNSAFE_SP_OFFSET with this value.
660 return SegmentOffset(IRB, 0x18, getAddressSpace());
661 }
662
664}
665
666//===----------------------------------------------------------------------===//
667// Return Value Calling Convention Implementation
668//===----------------------------------------------------------------------===//
669
670bool X86TargetLowering::CanLowerReturn(
671 CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg,
672 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
673 const Type *RetTy) const {
674 // Mingw64 GCC returns f128 via sret, and LLVM matches it for compatibility.
675 // This logic exists for libcalls, a frontend should explicitly use sret
676 // rather than rely on the sret demotion here.
677 //
678 // Using sret is a reasonable implementation of the Windows x64 calling
679 // convention:
680 //
681 // https://learn.microsoft.com/en-us/cpp/build/x64-calling-convention?view=msvc-170#return-values
682 //
683 // > Otherwise, the caller must allocate memory for the return value and pass
684 // > a pointer to it as the first argument.
685 //
686 // Although it is not the only reasonable interpretation:
687 //
688 // > Nonscalar types including floats, doubles, and vector types such as
689 // > __m128, __m128i, __m128d are returned in XMM0.
690 //
691 // For now, we prefer compatibility with GCC. If official guidelines are ever
692 // published, this can be revisited.
693 //
694 // Return false, which will perform sret demotion.
695 auto IsWin64F128StackCC = [this](CallingConv::ID CC) -> bool {
696 switch (CC) {
698 return true;
699 case CallingConv::C:
700 return Subtarget.isOSWindowsOrUEFI();
701 default:
702 return false;
703 }
704 };
705
706 if (IsWin64F128StackCC(CallConv) &&
708 Outs, [](const ISD::OutputArg &Out) { return Out.VT == MVT::f128; }))
709 return false;
710
712 CCState CCInfo(CallConv, isVarArg, MF, RVLocs, Context);
713 return CCInfo.CheckReturn(Outs, RetCC_X86);
714}
715
716const MCPhysReg *X86TargetLowering::getScratchRegisters(CallingConv::ID) const {
717 static const MCPhysReg ScratchRegs[] = { X86::R11, 0 };
718 return ScratchRegs;
719}
720
722 static const MCPhysReg RCRegs[] = {X86::FPCW, X86::MXCSR};
723 return RCRegs;
724}
725
726/// Lowers masks values (v*i1) to the local register values
727/// \returns DAG node after lowering to register type
728static SDValue lowerMasksToReg(const SDValue &ValArg, const EVT &ValLoc,
729 const SDLoc &DL, SelectionDAG &DAG) {
730 EVT ValVT = ValArg.getValueType();
731
732 if (ValVT == MVT::v1i1)
733 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ValLoc, ValArg,
734 DAG.getIntPtrConstant(0, DL));
735
736 if ((ValVT == MVT::v8i1 && (ValLoc == MVT::i8 || ValLoc == MVT::i32)) ||
737 (ValVT == MVT::v16i1 && (ValLoc == MVT::i16 || ValLoc == MVT::i32))) {
738 // Two stage lowering might be required
739 // bitcast: v8i1 -> i8 / v16i1 -> i16
740 // anyextend: i8 -> i32 / i16 -> i32
741 EVT TempValLoc = ValVT == MVT::v8i1 ? MVT::i8 : MVT::i16;
742 SDValue ValToCopy = DAG.getBitcast(TempValLoc, ValArg);
743 if (ValLoc == MVT::i32)
744 ValToCopy = DAG.getNode(ISD::ANY_EXTEND, DL, ValLoc, ValToCopy);
745 return ValToCopy;
746 }
747
748 if ((ValVT == MVT::v32i1 && ValLoc == MVT::i32) ||
749 (ValVT == MVT::v64i1 && ValLoc == MVT::i64)) {
750 // One stage lowering is required
751 // bitcast: v32i1 -> i32 / v64i1 -> i64
752 return DAG.getBitcast(ValLoc, ValArg);
753 }
754
755 return DAG.getNode(ISD::ANY_EXTEND, DL, ValLoc, ValArg);
756}
757
758/// Breaks v64i1 value into two registers and adds the new node to the DAG
760 const SDLoc &DL, SelectionDAG &DAG, SDValue &Arg,
761 SmallVectorImpl<std::pair<Register, SDValue>> &RegsToPass, CCValAssign &VA,
762 CCValAssign &NextVA, const X86Subtarget &Subtarget) {
763 assert(Subtarget.hasBWI() && "Expected AVX512BW target!");
764 assert(Subtarget.is32Bit() && "Expecting 32 bit target");
765 assert(Arg.getValueType() == MVT::i64 && "Expecting 64 bit value");
766 assert(VA.isRegLoc() && NextVA.isRegLoc() &&
767 "The value should reside in two registers");
768
769 // Before splitting the value we cast it to i64
770 Arg = DAG.getBitcast(MVT::i64, Arg);
771
772 // Splitting the value into two i32 types
773 SDValue Lo, Hi;
774 std::tie(Lo, Hi) = DAG.SplitScalar(Arg, DL, MVT::i32, MVT::i32);
775
776 // Attach the two i32 types into corresponding registers
777 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Lo));
778 RegsToPass.push_back(std::make_pair(NextVA.getLocReg(), Hi));
779}
780
782X86TargetLowering::LowerReturn(SDValue Chain, CallingConv::ID CallConv,
783 bool isVarArg,
785 const SmallVectorImpl<SDValue> &OutVals,
786 const SDLoc &dl, SelectionDAG &DAG) const {
787 MachineFunction &MF = DAG.getMachineFunction();
788 X86MachineFunctionInfo *FuncInfo = MF.getInfo<X86MachineFunctionInfo>();
789
790 // In some cases we need to disable registers from the default CSR list.
791 // For example, when they are used as return registers (preserve_* and X86's
792 // regcall) or for argument passing (X86's regcall).
793 bool ShouldDisableCalleeSavedRegister =
794 shouldDisableRetRegFromCSR(CallConv) ||
795 MF.getFunction().hasFnAttribute("no_caller_saved_registers");
796
797 if (CallConv == CallingConv::X86_INTR && !Outs.empty())
798 report_fatal_error("X86 interrupts may not return any value");
799
801 CCState CCInfo(CallConv, isVarArg, MF, RVLocs, *DAG.getContext());
802 CCInfo.AnalyzeReturn(Outs, RetCC_X86);
803
805 for (unsigned I = 0, OutsIndex = 0, E = RVLocs.size(); I != E;
806 ++I, ++OutsIndex) {
807 CCValAssign &VA = RVLocs[I];
808 assert(VA.isRegLoc() && "Can only return in registers!");
809
810 // Add the register to the CalleeSaveDisableRegs list.
811 if (ShouldDisableCalleeSavedRegister)
813
814 SDValue ValToCopy = OutVals[OutsIndex];
815 EVT ValVT = ValToCopy.getValueType();
816
817 // Promote values to the appropriate types.
818 if (VA.getLocInfo() == CCValAssign::SExt)
819 ValToCopy = DAG.getNode(ISD::SIGN_EXTEND, dl, VA.getLocVT(), ValToCopy);
820 else if (VA.getLocInfo() == CCValAssign::ZExt)
821 ValToCopy = DAG.getNode(ISD::ZERO_EXTEND, dl, VA.getLocVT(), ValToCopy);
822 else if (VA.getLocInfo() == CCValAssign::AExt) {
823 if (ValVT.isVectorOf(MVT::i1))
824 ValToCopy = lowerMasksToReg(ValToCopy, VA.getLocVT(), dl, DAG);
825 else
826 ValToCopy = DAG.getNode(ISD::ANY_EXTEND, dl, VA.getLocVT(), ValToCopy);
827 }
828 else if (VA.getLocInfo() == CCValAssign::BCvt)
829 ValToCopy = DAG.getBitcast(VA.getLocVT(), ValToCopy);
830
832 "Unexpected FP-extend for return value.");
833
834 // Report an error if we have attempted to return a value via an XMM
835 // register and SSE was disabled.
836 if (!Subtarget.hasSSE1() && X86::FR32XRegClass.contains(VA.getLocReg())) {
837 errorUnsupported(DAG, dl, "SSE register return with SSE disabled");
838 VA.convertToReg(X86::FP0); // Set reg to FP0, avoid hitting asserts.
839 } else if (!Subtarget.hasSSE2() &&
840 X86::FR64XRegClass.contains(VA.getLocReg()) &&
841 ValVT == MVT::f64) {
842 // When returning a double via an XMM register, report an error if SSE2 is
843 // not enabled.
844 errorUnsupported(DAG, dl, "SSE2 register return with SSE2 disabled");
845 VA.convertToReg(X86::FP0); // Set reg to FP0, avoid hitting asserts.
846 }
847
848 // Returns in ST0/ST1 are handled specially: these are pushed as operands to
849 // the RET instruction and handled by the FP Stackifier.
850 if (VA.getLocReg() == X86::FP0 ||
851 VA.getLocReg() == X86::FP1) {
852 // If this is a copy from an xmm register to ST(0), use an FPExtend to
853 // change the value to the FP stack register class.
855 ValToCopy = DAG.getNode(ISD::FP_EXTEND, dl, MVT::f80, ValToCopy);
856 RetVals.push_back(std::make_pair(VA.getLocReg(), ValToCopy));
857 // Don't emit a copytoreg.
858 continue;
859 }
860
861 // 64-bit vector (MMX) values are returned in XMM0 / XMM1 except for v1i64
862 // which is returned in RAX / RDX.
863 if (Subtarget.is64Bit()) {
864 if (ValVT == MVT::x86mmx) {
865 if (VA.getLocReg() == X86::XMM0 || VA.getLocReg() == X86::XMM1) {
866 ValToCopy = DAG.getBitcast(MVT::i64, ValToCopy);
867 ValToCopy = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2i64,
868 ValToCopy);
869 // If we don't have SSE2 available, convert to v4f32 so the generated
870 // register is legal.
871 if (!Subtarget.hasSSE2())
872 ValToCopy = DAG.getBitcast(MVT::v4f32, ValToCopy);
873 }
874 }
875 }
876
877 if (VA.needsCustom()) {
878 assert(VA.getValVT() == MVT::v64i1 &&
879 "Currently the only custom case is when we split v64i1 to 2 regs");
880
881 Passv64i1ArgInRegs(dl, DAG, ValToCopy, RetVals, VA, RVLocs[++I],
882 Subtarget);
883
884 // Add the second register to the CalleeSaveDisableRegs list.
885 if (ShouldDisableCalleeSavedRegister)
886 MF.getRegInfo().disableCalleeSavedRegister(RVLocs[I].getLocReg());
887 } else {
888 RetVals.push_back(std::make_pair(VA.getLocReg(), ValToCopy));
889 }
890 }
891
892 SDValue Glue;
894 RetOps.push_back(Chain); // Operand #0 = Chain (updated below)
895 // Operand #1 = Bytes To Pop
896 RetOps.push_back(DAG.getTargetConstant(FuncInfo->getBytesToPopOnReturn(), dl,
897 MVT::i32));
898
899 // Copy the result values into the output registers.
900 for (auto &RetVal : RetVals) {
901 if (RetVal.first == X86::FP0 || RetVal.first == X86::FP1) {
902 RetOps.push_back(RetVal.second);
903 continue; // Don't emit a copytoreg.
904 }
905
906 Chain = DAG.getCopyToReg(Chain, dl, RetVal.first, RetVal.second, Glue);
907 Glue = Chain.getValue(1);
908 RetOps.push_back(
909 DAG.getRegister(RetVal.first, RetVal.second.getValueType()));
910 }
911
912 // Swift calling convention does not require we copy the sret argument
913 // into %rax/%eax for the return, and SRetReturnReg is not set for Swift.
914
915 // All x86 ABIs require that for returning structs by value we copy
916 // the sret argument into %rax/%eax (depending on ABI) for the return.
917 // We saved the argument into a virtual register in the entry block,
918 // so now we copy the value out and into %rax/%eax.
919 //
920 // Checking Function.hasStructRetAttr() here is insufficient because the IR
921 // may not have an explicit sret argument. If FuncInfo.CanLowerReturn is
922 // false, then an sret argument may be implicitly inserted in the SelDAG. In
923 // either case FuncInfo->setSRetReturnReg() will have been called.
924 if (Register SRetReg = FuncInfo->getSRetReturnReg()) {
925 // When we have both sret and another return value, we should use the
926 // original Chain stored in RetOps[0], instead of the current Chain updated
927 // in the above loop. If we only have sret, RetOps[0] equals to Chain.
928
929 // For the case of sret and another return value, we have
930 // Chain_0 at the function entry
931 // Chain_1 = getCopyToReg(Chain_0) in the above loop
932 // If we use Chain_1 in getCopyFromReg, we will have
933 // Val = getCopyFromReg(Chain_1)
934 // Chain_2 = getCopyToReg(Chain_1, Val) from below
935
936 // getCopyToReg(Chain_0) will be glued together with
937 // getCopyToReg(Chain_1, Val) into Unit A, getCopyFromReg(Chain_1) will be
938 // in Unit B, and we will have cyclic dependency between Unit A and Unit B:
939 // Data dependency from Unit B to Unit A due to usage of Val in
940 // getCopyToReg(Chain_1, Val)
941 // Chain dependency from Unit A to Unit B
942
943 // So here, we use RetOps[0] (i.e Chain_0) for getCopyFromReg.
944 SDValue Val = DAG.getCopyFromReg(RetOps[0], dl, SRetReg,
946
947 Register RetValReg
948 = (Subtarget.is64Bit() && !Subtarget.isTarget64BitILP32()) ?
949 X86::RAX : X86::EAX;
950 Chain = DAG.getCopyToReg(Chain, dl, RetValReg, Val, Glue);
951 Glue = Chain.getValue(1);
952
953 // RAX/EAX now acts like a return value.
954 RetOps.push_back(
955 DAG.getRegister(RetValReg, getPointerTy(DAG.getDataLayout())));
956
957 // Add the returned register to the CalleeSaveDisableRegs list. Don't do
958 // this however for preserve_most/preserve_all to minimize the number of
959 // callee-saved registers for these CCs.
960 if (ShouldDisableCalleeSavedRegister &&
961 CallConv != CallingConv::PreserveAll &&
962 CallConv != CallingConv::PreserveMost)
964 }
965
966 const X86RegisterInfo *TRI = Subtarget.getRegisterInfo();
967 const MCPhysReg *I =
968 TRI->getCalleeSavedRegsViaCopy(&DAG.getMachineFunction());
969 if (I) {
970 for (; *I; ++I) {
971 if (X86::GR64RegClass.contains(*I))
972 RetOps.push_back(DAG.getRegister(*I, MVT::i64));
973 else
974 llvm_unreachable("Unexpected register class in CSRsViaCopy!");
975 }
976 }
977
978 RetOps[0] = Chain; // Update chain.
979
980 // Add the glue if we have it.
981 if (Glue.getNode())
982 RetOps.push_back(Glue);
983
984 unsigned RetOpcode = X86ISD::RET_GLUE;
985 if (CallConv == CallingConv::X86_INTR)
986 RetOpcode = X86ISD::IRET;
987 return DAG.getNode(RetOpcode, dl, MVT::Other, RetOps);
988}
989
990bool X86TargetLowering::isUsedByReturnOnly(SDNode *N, SDValue &Chain) const {
991 if (N->getNumValues() != 1 || !N->hasNUsesOfValue(1, 0))
992 return false;
993
994 SDValue TCChain = Chain;
995 SDNode *Copy = *N->user_begin();
996 if (Copy->getOpcode() == ISD::CopyToReg) {
997 // If the copy has a glue operand, we conservatively assume it isn't safe to
998 // perform a tail call.
999 if (Copy->getOperand(Copy->getNumOperands()-1).getValueType() == MVT::Glue)
1000 return false;
1001 TCChain = Copy->getOperand(0);
1002 } else if (Copy->getOpcode() != ISD::FP_EXTEND)
1003 return false;
1004
1005 bool HasRet = false;
1006 for (const SDNode *U : Copy->users()) {
1007 if (U->getOpcode() != X86ISD::RET_GLUE)
1008 return false;
1009 // If we are returning more than one value, we can definitely
1010 // not make a tail call see PR19530
1011 if (U->getNumOperands() > 4)
1012 return false;
1013 if (U->getNumOperands() == 4 &&
1014 U->getOperand(U->getNumOperands() - 1).getValueType() != MVT::Glue)
1015 return false;
1016 HasRet = true;
1017 }
1018
1019 if (!HasRet)
1020 return false;
1021
1022 Chain = TCChain;
1023 return true;
1024}
1025
1026EVT X86TargetLowering::getTypeForExtReturn(LLVMContext &Context, EVT VT,
1027 ISD::NodeType ExtendKind) const {
1028 MVT ReturnMVT = MVT::i32;
1029
1030 bool Darwin = Subtarget.getTargetTriple().isOSDarwin();
1031 if (VT == MVT::i1 || (!Darwin && (VT == MVT::i8 || VT == MVT::i16))) {
1032 // The ABI does not require i1, i8 or i16 to be extended.
1033 //
1034 // On Darwin, there is code in the wild relying on Clang's old behaviour of
1035 // always extending i8/i16 return values, so keep doing that for now.
1036 // (PR26665).
1037 ReturnMVT = MVT::i8;
1038 }
1039
1040 EVT MinVT = getRegisterType(Context, ReturnMVT);
1041 return VT.bitsLT(MinVT) ? MinVT : VT;
1042}
1043
1044/// Reads two 32 bit registers and creates a 64 bit mask value.
1045/// \param VA The current 32 bit value that need to be assigned.
1046/// \param NextVA The next 32 bit value that need to be assigned.
1047/// \param Root The parent DAG node.
1048/// \param [in,out] InGlue Represents SDvalue in the parent DAG node for
1049/// glue purposes. In the case the DAG is already using
1050/// physical register instead of virtual, we should glue
1051/// our new SDValue to InGlue SDvalue.
1052/// \return a new SDvalue of size 64bit.
1054 SDValue &Root, SelectionDAG &DAG,
1055 const SDLoc &DL, const X86Subtarget &Subtarget,
1056 SDValue *InGlue = nullptr) {
1057 assert((Subtarget.hasBWI()) && "Expected AVX512BW target!");
1058 assert(Subtarget.is32Bit() && "Expecting 32 bit target");
1059 assert(VA.getValVT() == MVT::v64i1 &&
1060 "Expecting first location of 64 bit width type");
1061 assert(NextVA.getValVT() == VA.getValVT() &&
1062 "The locations should have the same type");
1063 assert(VA.isRegLoc() && NextVA.isRegLoc() &&
1064 "The values should reside in two registers");
1065
1066 SDValue Lo, Hi;
1067 SDValue ArgValueLo, ArgValueHi;
1068
1070 const TargetRegisterClass *RC = &X86::GR32RegClass;
1071
1072 // Read a 32 bit value from the registers.
1073 if (nullptr == InGlue) {
1074 // When no physical register is present,
1075 // create an intermediate virtual register.
1076 Register Reg = MF.addLiveIn(VA.getLocReg(), RC);
1077 ArgValueLo = DAG.getCopyFromReg(Root, DL, Reg, MVT::i32);
1078 Reg = MF.addLiveIn(NextVA.getLocReg(), RC);
1079 ArgValueHi = DAG.getCopyFromReg(Root, DL, Reg, MVT::i32);
1080 } else {
1081 // When a physical register is available read the value from it and glue
1082 // the reads together.
1083 ArgValueLo =
1084 DAG.getCopyFromReg(Root, DL, VA.getLocReg(), MVT::i32, *InGlue);
1085 *InGlue = ArgValueLo.getValue(2);
1086 ArgValueHi =
1087 DAG.getCopyFromReg(Root, DL, NextVA.getLocReg(), MVT::i32, *InGlue);
1088 *InGlue = ArgValueHi.getValue(2);
1089 }
1090
1091 // Convert the i32 type into v32i1 type.
1092 Lo = DAG.getBitcast(MVT::v32i1, ArgValueLo);
1093
1094 // Convert the i32 type into v32i1 type.
1095 Hi = DAG.getBitcast(MVT::v32i1, ArgValueHi);
1096
1097 // Concatenate the two values together.
1098 return DAG.getNode(ISD::CONCAT_VECTORS, DL, MVT::v64i1, Lo, Hi);
1099}
1100
1101/// The function will lower a register of various sizes (8/16/32/64)
1102/// to a mask value of the expected size (v8i1/v16i1/v32i1/v64i1)
1103/// \returns a DAG node contains the operand after lowering to mask type.
1104static SDValue lowerRegToMasks(const SDValue &ValArg, const EVT &ValVT,
1105 const EVT &ValLoc, const SDLoc &DL,
1106 SelectionDAG &DAG) {
1107 SDValue ValReturned = ValArg;
1108
1109 if (ValVT == MVT::v1i1)
1110 return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, MVT::v1i1, ValReturned);
1111
1112 if (ValVT == MVT::v64i1) {
1113 // In 32 bit machine, this case is handled by getv64i1Argument
1114 assert(ValLoc == MVT::i64 && "Expecting only i64 locations");
1115 // In 64 bit machine, There is no need to truncate the value only bitcast
1116 } else {
1117 MVT MaskLenVT;
1118 switch (ValVT.getSimpleVT().SimpleTy) {
1119 case MVT::v8i1:
1120 MaskLenVT = MVT::i8;
1121 break;
1122 case MVT::v16i1:
1123 MaskLenVT = MVT::i16;
1124 break;
1125 case MVT::v32i1:
1126 MaskLenVT = MVT::i32;
1127 break;
1128 default:
1129 llvm_unreachable("Expecting a vector of i1 types");
1130 }
1131
1132 ValReturned = DAG.getNode(ISD::TRUNCATE, DL, MaskLenVT, ValReturned);
1133 }
1134 return DAG.getBitcast(ValVT, ValReturned);
1135}
1136
1138 const SDLoc &dl, Register Reg, EVT VT,
1139 SDValue Glue) {
1140 SDVTList VTs = DAG.getVTList(VT, MVT::Other, MVT::Glue);
1141 SDValue Ops[] = {Chain, DAG.getRegister(Reg, VT), Glue};
1142 return DAG.getNode(X86ISD::POP_FROM_X87_REG, dl, VTs,
1143 ArrayRef(Ops, Glue.getNode() ? 3 : 2));
1144}
1145
1146/// Lower the result values of a call into the
1147/// appropriate copies out of appropriate physical registers.
1148///
1149SDValue X86TargetLowering::LowerCallResult(
1150 SDValue Chain, SDValue InGlue, CallingConv::ID CallConv, bool isVarArg,
1151 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl,
1153 uint32_t *RegMask) const {
1154
1155 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
1156 // Assign locations to each value returned by this call.
1158 CCState CCInfo(CallConv, isVarArg, DAG.getMachineFunction(), RVLocs,
1159 *DAG.getContext());
1160 CCInfo.AnalyzeCallResult(Ins, RetCC_X86);
1161
1162 // Copy all of the result registers out of their specified physreg.
1163 for (unsigned I = 0, InsIndex = 0, E = RVLocs.size(); I != E;
1164 ++I, ++InsIndex) {
1165 CCValAssign &VA = RVLocs[I];
1166 EVT CopyVT = VA.getLocVT();
1167
1168 // In some calling conventions we need to remove the used registers
1169 // from the register mask.
1170 if (RegMask) {
1171 for (MCPhysReg SubReg : TRI->subregs_inclusive(VA.getLocReg()))
1172 RegMask[SubReg / 32] &= ~(1u << (SubReg % 32));
1173 }
1174
1175 // Report an error if there was an attempt to return FP values via XMM
1176 // registers.
1177 if (!Subtarget.hasSSE1() && X86::FR32XRegClass.contains(VA.getLocReg())) {
1178 errorUnsupported(DAG, dl, "SSE register return with SSE disabled");
1179 if (VA.getLocReg() == X86::XMM1)
1180 VA.convertToReg(X86::FP1); // Set reg to FP1, avoid hitting asserts.
1181 else
1182 VA.convertToReg(X86::FP0); // Set reg to FP0, avoid hitting asserts.
1183 } else if (!Subtarget.hasSSE2() &&
1184 X86::FR64XRegClass.contains(VA.getLocReg()) &&
1185 CopyVT == MVT::f64) {
1186 errorUnsupported(DAG, dl, "SSE2 register return with SSE2 disabled");
1187 if (VA.getLocReg() == X86::XMM1)
1188 VA.convertToReg(X86::FP1); // Set reg to FP1, avoid hitting asserts.
1189 else
1190 VA.convertToReg(X86::FP0); // Set reg to FP0, avoid hitting asserts.
1191 }
1192
1193 // If we prefer to use the value in xmm registers, copy it out as f80 and
1194 // use a truncate to move it from fp stack reg to xmm reg.
1195 bool RoundAfterCopy = false;
1196 bool X87Result = VA.getLocReg() == X86::FP0 || VA.getLocReg() == X86::FP1;
1197 if (X87Result && isScalarFPTypeInSSEReg(VA.getValVT())) {
1198 if (!Subtarget.hasX87())
1199 report_fatal_error("X87 register return with X87 disabled");
1200 CopyVT = MVT::f80;
1201 RoundAfterCopy = (CopyVT != VA.getLocVT());
1202 }
1203
1204 SDValue Val;
1205 if (VA.needsCustom()) {
1206 assert(VA.getValVT() == MVT::v64i1 &&
1207 "Currently the only custom case is when we split v64i1 to 2 regs");
1208 Val =
1209 getv64i1Argument(VA, RVLocs[++I], Chain, DAG, dl, Subtarget, &InGlue);
1210 } else {
1211 Chain =
1212 X87Result
1213 ? getPopFromX87Reg(DAG, Chain, dl, VA.getLocReg(), CopyVT, InGlue)
1214 .getValue(1)
1215 : DAG.getCopyFromReg(Chain, dl, VA.getLocReg(), CopyVT, InGlue)
1216 .getValue(1);
1217 Val = Chain.getValue(0);
1218 InGlue = Chain.getValue(2);
1219 }
1220
1221 if (RoundAfterCopy)
1222 Val = DAG.getNode(ISD::FP_ROUND, dl, VA.getValVT(), Val,
1223 // This truncation won't change the value.
1224 DAG.getIntPtrConstant(1, dl, /*isTarget=*/true));
1225
1226 if (VA.isExtInLoc()) {
1227 if (VA.getValVT().isVector() &&
1228 VA.getValVT().getScalarType() == MVT::i1 &&
1229 ((VA.getLocVT() == MVT::i64) || (VA.getLocVT() == MVT::i32) ||
1230 (VA.getLocVT() == MVT::i16) || (VA.getLocVT() == MVT::i8))) {
1231 // promoting a mask type (v*i1) into a register of type i64/i32/i16/i8
1232 Val = lowerRegToMasks(Val, VA.getValVT(), VA.getLocVT(), dl, DAG);
1233 } else
1234 Val = DAG.getNode(ISD::TRUNCATE, dl, VA.getValVT(), Val);
1235 }
1236
1237 if (VA.getLocInfo() == CCValAssign::BCvt)
1238 Val = DAG.getBitcast(VA.getValVT(), Val);
1239
1240 InVals.push_back(Val);
1241 }
1242
1243 return Chain;
1244}
1245
1246/// Determines whether Args, either a set of outgoing arguments to a call, or a
1247/// set of incoming args of a call, contains an sret pointer that the callee
1248/// pops. This happens on most x86-32, System V platforms, unless register
1249/// parameters are in use (-mregparm=1+, regcallcc, etc).
1250template <typename T>
1251static bool hasCalleePopSRet(const SmallVectorImpl<T> &Args,
1252 const SmallVectorImpl<CCValAssign> &ArgLocs,
1253 const X86Subtarget &Subtarget) {
1254 // Not C++20 (yet), so no concepts available.
1255 static_assert(std::is_same_v<T, ISD::OutputArg> ||
1256 std::is_same_v<T, ISD::InputArg>,
1257 "requires ISD::OutputArg or ISD::InputArg");
1258
1259 // Popping the sret pointer only happens on x86-32 System V ABI platforms
1260 // (Linux, Cygwin, BSDs, Mac, etc). That excludes Windows-minus-Cygwin and
1261 // MCU.
1262 const Triple &TT = Subtarget.getTargetTriple();
1263 if (!TT.isX86_32() || TT.isOSMSVCRT() || TT.isOSIAMCU())
1264 return false;
1265
1266 // Check if the first argument is marked sret and if it is passed in memory.
1267 bool IsSRetInMem = false;
1268 if (!Args.empty())
1269 IsSRetInMem = Args.front().Flags.isSRet() && ArgLocs.front().isMemLoc();
1270 return IsSRetInMem;
1271}
1272
1273/// Make a copy of an aggregate at address specified by "Src" to address
1274/// "Dst" with size and alignment information specified by the specific
1275/// parameter attribute. The copy will be passed as a byval function parameter.
1277 SDValue Chain, ISD::ArgFlagsTy Flags,
1278 SelectionDAG &DAG, const SDLoc &dl) {
1279 SDValue SizeNode = DAG.getIntPtrConstant(Flags.getByValSize(), dl);
1280 Align Alignment = Flags.getNonZeroByValAlign();
1281 return DAG.getMemcpy(Chain, dl, Dst, Src, SizeNode, Alignment, Alignment,
1282 /*isVolatile*/ false, /*AlwaysInline=*/true,
1283 /*CI=*/nullptr, std::nullopt, MachinePointerInfo(),
1285}
1286
1287/// Return true if the calling convention is one that we can guarantee TCO for.
1289 return (CC == CallingConv::Fast || CC == CallingConv::GHC ||
1292}
1293
1294/// Return true if we might ever do TCO for calls with this calling convention.
1296 switch (CC) {
1297 // C calling conventions:
1298 case CallingConv::C:
1299 case CallingConv::Win64:
1302 // Callee pop conventions:
1307 // Swift:
1308 case CallingConv::Swift:
1309 return true;
1310 default:
1311 return canGuaranteeTCO(CC);
1312 }
1313}
1314
1315/// Return true if the function is being made into a tailcall target by
1316/// changing its ABI.
1317static bool shouldGuaranteeTCO(CallingConv::ID CC, bool GuaranteedTailCallOpt) {
1318 return (GuaranteedTailCallOpt && canGuaranteeTCO(CC)) ||
1320}
1321
1322bool X86TargetLowering::mayBeEmittedAsTailCall(const CallInst *CI) const {
1323 if (!CI->isTailCall())
1324 return false;
1325
1326 CallingConv::ID CalleeCC = CI->getCallingConv();
1327 if (!mayTailCallThisCC(CalleeCC))
1328 return false;
1329
1330 return true;
1331}
1332
1333SDValue
1334X86TargetLowering::LowerMemArgument(SDValue Chain, CallingConv::ID CallConv,
1336 const SDLoc &dl, SelectionDAG &DAG,
1337 const CCValAssign &VA,
1338 MachineFrameInfo &MFI, unsigned i) const {
1339 // Create the nodes corresponding to a load from this parameter slot.
1340 ISD::ArgFlagsTy Flags = Ins[i].Flags;
1341 bool AlwaysUseMutable = shouldGuaranteeTCO(
1342 CallConv, DAG.getTarget().Options.GuaranteedTailCallOpt);
1343 bool isImmutable = !AlwaysUseMutable && !Flags.isByVal();
1344 EVT ValVT;
1345 MVT PtrVT = getPointerTy(DAG.getDataLayout());
1346
1347 // If value is passed by pointer we have address passed instead of the value
1348 // itself. No need to extend if the mask value and location share the same
1349 // absolute size.
1350 bool ExtendedInMem =
1351 VA.isExtInLoc() && VA.getValVT().getScalarType() == MVT::i1 &&
1353
1354 if (VA.getLocInfo() == CCValAssign::Indirect || ExtendedInMem)
1355 ValVT = VA.getLocVT();
1356 else
1357 ValVT = VA.getValVT();
1358
1359 // FIXME: For now, all byval parameter objects are marked mutable. This can be
1360 // changed with more analysis.
1361 // In case of tail call optimization mark all arguments mutable. Since they
1362 // could be overwritten by lowering of arguments in case of a tail call.
1363 if (Flags.isByVal()) {
1364 unsigned Bytes = Flags.getByValSize();
1365 if (Bytes == 0) Bytes = 1; // Don't create zero-sized stack objects.
1366
1367 // FIXME: For now, all byval parameter objects are marked as aliasing. This
1368 // can be improved with deeper analysis.
1369 int FI = MFI.CreateFixedObject(Bytes, VA.getLocMemOffset(), isImmutable,
1370 /*isAliased=*/true);
1371 return DAG.getFrameIndex(FI, PtrVT);
1372 }
1373
1374 EVT ArgVT = Ins[i].ArgVT;
1375
1376 // If this is a vector that has been split into multiple parts, don't elide
1377 // the copy. The layout on the stack may not match the packed in-memory
1378 // layout.
1379 bool ScalarizedVector = ArgVT.isVector() && !VA.getLocVT().isVector();
1380
1381 // This is an argument in memory. We might be able to perform copy elision.
1382 // If the argument is passed directly in memory without any extension, then we
1383 // can perform copy elision. Large vector types, for example, may be passed
1384 // indirectly by pointer.
1385 if (Flags.isCopyElisionCandidate() &&
1386 VA.getLocInfo() != CCValAssign::Indirect && !ExtendedInMem &&
1387 !ScalarizedVector) {
1388 SDValue PartAddr;
1389 if (Ins[i].PartOffset == 0) {
1390 // If this is a one-part value or the first part of a multi-part value,
1391 // create a stack object for the entire argument value type and return a
1392 // load from our portion of it. This assumes that if the first part of an
1393 // argument is in memory, the rest will also be in memory.
1394 int FI = MFI.CreateFixedObject(ArgVT.getStoreSize(), VA.getLocMemOffset(),
1395 /*IsImmutable=*/false);
1396 PartAddr = DAG.getFrameIndex(FI, PtrVT);
1397 return DAG.getLoad(
1398 ValVT, dl, Chain, PartAddr,
1400 }
1401
1402 // This is not the first piece of an argument in memory. See if there is
1403 // already a fixed stack object including this offset. If so, assume it
1404 // was created by the PartOffset == 0 branch above and create a load from
1405 // the appropriate offset into it.
1406 int64_t PartBegin = VA.getLocMemOffset();
1407 int64_t PartEnd = PartBegin + ValVT.getSizeInBits() / 8;
1408 int FI = MFI.getObjectIndexBegin();
1409 for (; MFI.isFixedObjectIndex(FI); ++FI) {
1410 int64_t ObjBegin = MFI.getObjectOffset(FI);
1411 int64_t ObjEnd = ObjBegin + MFI.getObjectSize(FI);
1412 if (ObjBegin <= PartBegin && PartEnd <= ObjEnd)
1413 break;
1414 }
1415 if (MFI.isFixedObjectIndex(FI)) {
1416 SDValue Addr =
1417 DAG.getNode(ISD::ADD, dl, PtrVT, DAG.getFrameIndex(FI, PtrVT),
1418 DAG.getIntPtrConstant(Ins[i].PartOffset, dl));
1419 return DAG.getLoad(ValVT, dl, Chain, Addr,
1421 DAG.getMachineFunction(), FI, Ins[i].PartOffset));
1422 }
1423 }
1424
1425 int FI = MFI.CreateFixedObject(ValVT.getSizeInBits() / 8,
1426 VA.getLocMemOffset(), isImmutable);
1427
1428 // Set SExt or ZExt flag.
1429 if (VA.getLocInfo() == CCValAssign::ZExt) {
1430 MFI.setObjectZExt(FI, true);
1431 } else if (VA.getLocInfo() == CCValAssign::SExt) {
1432 MFI.setObjectSExt(FI, true);
1433 }
1434
1435 MaybeAlign Alignment;
1436 if (Subtarget.isTargetWindowsMSVC() && !Subtarget.is64Bit() &&
1437 ValVT != MVT::f80)
1438 Alignment = MaybeAlign(4);
1439 SDValue FIN = DAG.getFrameIndex(FI, PtrVT);
1440 SDValue Val = DAG.getLoad(
1441 ValVT, dl, Chain, FIN,
1443 Alignment);
1444 return ExtendedInMem
1445 ? (VA.getValVT().isVector()
1446 ? DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, VA.getValVT(), Val)
1447 : DAG.getNode(ISD::TRUNCATE, dl, VA.getValVT(), Val))
1448 : Val;
1449}
1450
1451// FIXME: Get this from tablegen.
1453 const X86Subtarget &Subtarget) {
1454 assert(Subtarget.is64Bit());
1455
1456 if (Subtarget.isCallingConvWin64(CallConv)) {
1457 static const MCPhysReg GPR64ArgRegsWin64[] = {
1458 X86::RCX, X86::RDX, X86::R8, X86::R9
1459 };
1460 return GPR64ArgRegsWin64;
1461 }
1462
1463 static const MCPhysReg GPR64ArgRegs64Bit[] = {
1464 X86::RDI, X86::RSI, X86::RDX, X86::RCX, X86::R8, X86::R9
1465 };
1466 return GPR64ArgRegs64Bit;
1467}
1468
1469// FIXME: Get this from tablegen.
1471 CallingConv::ID CallConv,
1472 const X86Subtarget &Subtarget) {
1473 assert(Subtarget.is64Bit());
1474 if (Subtarget.isCallingConvWin64(CallConv)) {
1475 // The XMM registers which might contain var arg parameters are shadowed
1476 // in their paired GPR. So we only need to save the GPR to their home
1477 // slots.
1478 // TODO: __vectorcall will change this.
1479 return {};
1480 }
1481
1482 bool isSoftFloat = Subtarget.useSoftFloat();
1483 if (isSoftFloat || !Subtarget.hasSSE1())
1484 // Kernel mode asks for SSE to be disabled, so there are no XMM argument
1485 // registers.
1486 return {};
1487
1488 static const MCPhysReg XMMArgRegs64Bit[] = {
1489 X86::XMM0, X86::XMM1, X86::XMM2, X86::XMM3,
1490 X86::XMM4, X86::XMM5, X86::XMM6, X86::XMM7
1491 };
1492 return XMMArgRegs64Bit;
1493}
1494
1495#ifndef NDEBUG
1497 return llvm::is_sorted(
1498 ArgLocs, [](const CCValAssign &A, const CCValAssign &B) -> bool {
1499 return A.getValNo() < B.getValNo();
1500 });
1501}
1502#endif
1503
1504namespace {
1505/// This is a helper class for lowering variable arguments parameters.
1506class VarArgsLoweringHelper {
1507public:
1508 VarArgsLoweringHelper(X86MachineFunctionInfo *FuncInfo, const SDLoc &Loc,
1509 SelectionDAG &DAG, const X86Subtarget &Subtarget,
1510 CallingConv::ID CallConv, CCState &CCInfo)
1511 : FuncInfo(FuncInfo), DL(Loc), DAG(DAG), Subtarget(Subtarget),
1512 TheMachineFunction(DAG.getMachineFunction()),
1513 TheFunction(TheMachineFunction.getFunction()),
1514 FrameInfo(TheMachineFunction.getFrameInfo()),
1515 FrameLowering(*Subtarget.getFrameLowering()),
1516 TargLowering(DAG.getTargetLoweringInfo()), CallConv(CallConv),
1517 CCInfo(CCInfo) {}
1518
1519 // Lower variable arguments parameters.
1520 void lowerVarArgsParameters(SDValue &Chain, unsigned StackSize);
1521
1522private:
1523 void createVarArgAreaAndStoreRegisters(SDValue &Chain, unsigned StackSize);
1524
1525 void forwardMustTailParameters(SDValue &Chain);
1526
1527 bool is64Bit() const { return Subtarget.is64Bit(); }
1528 bool isWin64() const { return Subtarget.isCallingConvWin64(CallConv); }
1529
1530 X86MachineFunctionInfo *FuncInfo;
1531 const SDLoc &DL;
1532 SelectionDAG &DAG;
1533 const X86Subtarget &Subtarget;
1534 MachineFunction &TheMachineFunction;
1535 const Function &TheFunction;
1536 MachineFrameInfo &FrameInfo;
1537 const TargetFrameLowering &FrameLowering;
1538 const TargetLowering &TargLowering;
1539 CallingConv::ID CallConv;
1540 CCState &CCInfo;
1541};
1542} // namespace
1543
1544void VarArgsLoweringHelper::createVarArgAreaAndStoreRegisters(
1545 SDValue &Chain, unsigned StackSize) {
1546 // If the function takes variable number of arguments, make a frame index for
1547 // the start of the first vararg value... for expansion of llvm.va_start. We
1548 // can skip this if there are no va_start calls.
1549 if (is64Bit() || (CallConv != CallingConv::X86_FastCall &&
1550 CallConv != CallingConv::X86_ThisCall)) {
1551 FuncInfo->setVarArgsFrameIndex(
1552 FrameInfo.CreateFixedObject(1, StackSize, true));
1553 }
1554
1555 // 64-bit calling conventions support varargs and register parameters, so we
1556 // have to do extra work to spill them in the prologue.
1557 if (is64Bit()) {
1558 // Find the first unallocated argument registers.
1559 ArrayRef<MCPhysReg> ArgGPRs = get64BitArgumentGPRs(CallConv, Subtarget);
1560 ArrayRef<MCPhysReg> ArgXMMs =
1561 get64BitArgumentXMMs(TheMachineFunction, CallConv, Subtarget);
1562 unsigned NumIntRegs = CCInfo.getFirstUnallocated(ArgGPRs);
1563 unsigned NumXMMRegs = CCInfo.getFirstUnallocated(ArgXMMs);
1564
1565 assert(!(NumXMMRegs && !Subtarget.hasSSE1()) &&
1566 "SSE register cannot be used when SSE is disabled!");
1567
1568 if (isWin64()) {
1569 // Get to the caller-allocated home save location. Add 8 to account
1570 // for the return address.
1571 int HomeOffset = FrameLowering.getOffsetOfLocalArea() + 8;
1572 FuncInfo->setRegSaveFrameIndex(
1573 FrameInfo.CreateFixedObject(1, NumIntRegs * 8 + HomeOffset, false));
1574 // Fixup to set vararg frame on shadow area (4 x i64).
1575 if (NumIntRegs < 4)
1576 FuncInfo->setVarArgsFrameIndex(FuncInfo->getRegSaveFrameIndex());
1577 } else {
1578 // For X86-64, if there are vararg parameters that are passed via
1579 // registers, then we must store them to their spots on the stack so
1580 // they may be loaded by dereferencing the result of va_next.
1581 FuncInfo->setVarArgsGPOffset(NumIntRegs * 8);
1582 FuncInfo->setVarArgsFPOffset(ArgGPRs.size() * 8 + NumXMMRegs * 16);
1583 FuncInfo->setRegSaveFrameIndex(FrameInfo.CreateStackObject(
1584 ArgGPRs.size() * 8 + ArgXMMs.size() * 16, Align(16), false));
1585 }
1586
1588 LiveGPRs; // list of SDValue for GPR registers keeping live input value
1589 SmallVector<SDValue, 8> LiveXMMRegs; // list of SDValue for XMM registers
1590 // keeping live input value
1591 SDValue ALVal; // if applicable keeps SDValue for %al register
1592
1593 // Gather all the live in physical registers.
1594 for (MCPhysReg Reg : ArgGPRs.slice(NumIntRegs)) {
1595 Register GPR = TheMachineFunction.addLiveIn(Reg, &X86::GR64RegClass);
1596 LiveGPRs.push_back(DAG.getCopyFromReg(Chain, DL, GPR, MVT::i64));
1597 }
1598 const auto &AvailableXmms = ArgXMMs.slice(NumXMMRegs);
1599 if (!AvailableXmms.empty()) {
1600 Register AL = TheMachineFunction.addLiveIn(X86::AL, &X86::GR8RegClass);
1601 ALVal = DAG.getCopyFromReg(Chain, DL, AL, MVT::i8);
1602 for (MCPhysReg Reg : AvailableXmms) {
1603 // FastRegisterAllocator spills virtual registers at basic
1604 // block boundary. That leads to usages of xmm registers
1605 // outside of check for %al. Pass physical registers to
1606 // VASTART_SAVE_XMM_REGS to avoid unneccessary spilling.
1607 TheMachineFunction.getRegInfo().addLiveIn(Reg);
1608 LiveXMMRegs.push_back(DAG.getRegister(Reg, MVT::v4f32));
1609 }
1610 }
1611
1612 // Store the integer parameter registers.
1614 SDValue RSFIN =
1615 DAG.getFrameIndex(FuncInfo->getRegSaveFrameIndex(),
1616 TargLowering.getPointerTy(DAG.getDataLayout()));
1617 unsigned Offset = FuncInfo->getVarArgsGPOffset();
1618 for (SDValue Val : LiveGPRs) {
1619 SDValue FIN = DAG.getNode(ISD::ADD, DL,
1620 TargLowering.getPointerTy(DAG.getDataLayout()),
1621 RSFIN, DAG.getIntPtrConstant(Offset, DL));
1622 SDValue Store =
1623 DAG.getStore(Val.getValue(1), DL, Val, FIN,
1625 DAG.getMachineFunction(),
1626 FuncInfo->getRegSaveFrameIndex(), Offset));
1627 MemOps.push_back(Store);
1628 Offset += 8;
1629 }
1630
1631 // Now store the XMM (fp + vector) parameter registers.
1632 if (!LiveXMMRegs.empty()) {
1633 SmallVector<SDValue, 12> SaveXMMOps;
1634 SaveXMMOps.push_back(Chain);
1635 SaveXMMOps.push_back(ALVal);
1636 SaveXMMOps.push_back(RSFIN);
1637 SaveXMMOps.push_back(
1638 DAG.getTargetConstant(FuncInfo->getVarArgsFPOffset(), DL, MVT::i32));
1639 llvm::append_range(SaveXMMOps, LiveXMMRegs);
1640 MachineMemOperand *StoreMMO =
1643 DAG.getMachineFunction(), FuncInfo->getRegSaveFrameIndex(),
1644 Offset),
1646 MemOps.push_back(DAG.getMemIntrinsicNode(X86ISD::VASTART_SAVE_XMM_REGS,
1647 DL, DAG.getVTList(MVT::Other),
1648 SaveXMMOps, MVT::i8, StoreMMO));
1649 }
1650
1651 if (!MemOps.empty())
1652 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOps);
1653 }
1654}
1655
1656void VarArgsLoweringHelper::forwardMustTailParameters(SDValue &Chain) {
1657 // Find the largest legal vector type.
1658 MVT VecVT = MVT::Other;
1659 // FIXME: Only some x86_32 calling conventions support AVX512.
1660 if (Subtarget.useAVX512Regs() &&
1661 (is64Bit() || (CallConv == CallingConv::X86_VectorCall ||
1662 CallConv == CallingConv::Intel_OCL_BI)))
1663 VecVT = MVT::v16f32;
1664 else if (Subtarget.hasAVX())
1665 VecVT = MVT::v8f32;
1666 else if (Subtarget.hasSSE2())
1667 VecVT = MVT::v4f32;
1668
1669 // We forward some GPRs and some vector types.
1670 SmallVector<MVT, 2> RegParmTypes;
1671 MVT IntVT = is64Bit() ? MVT::i64 : MVT::i32;
1672 RegParmTypes.push_back(IntVT);
1673 if (VecVT != MVT::Other)
1674 RegParmTypes.push_back(VecVT);
1675
1676 // Compute the set of forwarded registers. The rest are scratch.
1678 FuncInfo->getForwardedMustTailRegParms();
1679 CCInfo.analyzeMustTailForwardedRegisters(Forwards, RegParmTypes, CC_X86);
1680
1681 // Forward AL for SysV x86_64 targets, since it is used for varargs.
1682 if (is64Bit() && !isWin64() && !CCInfo.isAllocated(X86::AL)) {
1683 Register ALVReg = TheMachineFunction.addLiveIn(X86::AL, &X86::GR8RegClass);
1684 Forwards.push_back(ForwardedRegister(ALVReg, X86::AL, MVT::i8));
1685 }
1686
1687 // Copy all forwards from physical to virtual registers.
1688 for (ForwardedRegister &FR : Forwards) {
1689 // FIXME: Can we use a less constrained schedule?
1690 SDValue RegVal = DAG.getCopyFromReg(Chain, DL, FR.VReg, FR.VT);
1691 FR.VReg = TheMachineFunction.getRegInfo().createVirtualRegister(
1692 TargLowering.getRegClassFor(FR.VT));
1693 Chain = DAG.getCopyToReg(Chain, DL, FR.VReg, RegVal);
1694 }
1695}
1696
1697void VarArgsLoweringHelper::lowerVarArgsParameters(SDValue &Chain,
1698 unsigned StackSize) {
1699 // Set FrameIndex to the 0xAAAAAAA value to mark unset state.
1700 // If necessary, it would be set into the correct value later.
1701 FuncInfo->setVarArgsFrameIndex(0xAAAAAAA);
1702 FuncInfo->setRegSaveFrameIndex(0xAAAAAAA);
1703
1704 if (FrameInfo.hasVAStart())
1705 createVarArgAreaAndStoreRegisters(Chain, StackSize);
1706
1707 if (FrameInfo.hasMustTailInVarArgFunc())
1708 forwardMustTailParameters(Chain);
1709}
1710
1711SDValue X86TargetLowering::LowerFormalArguments(
1712 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
1713 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &dl,
1714 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
1715 MachineFunction &MF = DAG.getMachineFunction();
1716 X86MachineFunctionInfo *FuncInfo = MF.getInfo<X86MachineFunctionInfo>();
1717
1718 const Function &F = MF.getFunction();
1719 if (F.hasExternalLinkage() && Subtarget.isTargetCygMing() &&
1720 F.getName() == "main")
1721 FuncInfo->setForceFramePointer(true);
1722
1723 MachineFrameInfo &MFI = MF.getFrameInfo();
1724 bool Is64Bit = Subtarget.is64Bit();
1725 bool IsWin64 = Subtarget.isCallingConvWin64(CallConv);
1726
1727 // On x86_64 with x87 disabled, x86_fp80 cannot be handled: the type would
1728 // need to be returned/passed in x87 registers (FP0/FP1) which are
1729 // unavailable. Emit a clear diagnostic instead of crashing later with
1730 // "Cannot select: build_pair".
1731 if (Is64Bit && !Subtarget.hasX87()) {
1732 if (F.getReturnType()->isX86_FP80Ty() ||
1733 any_of(F.args(), [](const Argument &Arg) {
1734 return Arg.getType()->isX86_FP80Ty();
1735 }))
1737 "cannot use x86_fp80 type with x87 disabled on x86_64 target");
1738 }
1739
1740 assert(
1741 !(IsVarArg && canGuaranteeTCO(CallConv)) &&
1742 "Var args not supported with calling conv' regcall, fastcc, ghc or hipe");
1743
1744 // Assign locations to all of the incoming arguments.
1746 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
1747
1748 // Allocate shadow area for Win64.
1749 if (IsWin64)
1750 CCInfo.AllocateStack(32, Align(8));
1751
1752 CCInfo.AnalyzeArguments(Ins, CC_X86);
1753
1754 // In vectorcall calling convention a second pass is required for the HVA
1755 // types.
1756 if (CallingConv::X86_VectorCall == CallConv) {
1757 CCInfo.AnalyzeArgumentsSecondPass(Ins, CC_X86);
1758 }
1759
1760 // The next loop assumes that the locations are in the same order of the
1761 // input arguments.
1762 assert(isSortedByValueNo(ArgLocs) &&
1763 "Argument Location list must be sorted before lowering");
1764
1765 SDValue ArgValue;
1766 for (unsigned I = 0, InsIndex = 0, E = ArgLocs.size(); I != E;
1767 ++I, ++InsIndex) {
1768 assert(InsIndex < Ins.size() && "Invalid Ins index");
1769 CCValAssign &VA = ArgLocs[I];
1770
1771 if (VA.isRegLoc()) {
1772 EVT RegVT = VA.getLocVT();
1773 if (VA.needsCustom()) {
1774 assert(
1775 VA.getValVT() == MVT::v64i1 &&
1776 "Currently the only custom case is when we split v64i1 to 2 regs");
1777
1778 // v64i1 values, in regcall calling convention, that are
1779 // compiled to 32 bit arch, are split up into two registers.
1780 ArgValue =
1781 getv64i1Argument(VA, ArgLocs[++I], Chain, DAG, dl, Subtarget);
1782 } else {
1783 const TargetRegisterClass *RC;
1784 if (RegVT == MVT::i8)
1785 RC = &X86::GR8RegClass;
1786 else if (RegVT == MVT::i16)
1787 RC = &X86::GR16RegClass;
1788 else if (RegVT == MVT::i32)
1789 RC = &X86::GR32RegClass;
1790 else if (Is64Bit && RegVT == MVT::i64)
1791 RC = &X86::GR64RegClass;
1792 else if (RegVT == MVT::f16)
1793 RC = Subtarget.hasAVX512() ? &X86::FR16XRegClass : &X86::FR16RegClass;
1794 else if (RegVT == MVT::f32)
1795 RC = Subtarget.hasAVX512() ? &X86::FR32XRegClass : &X86::FR32RegClass;
1796 else if (RegVT == MVT::f64)
1797 RC = Subtarget.hasAVX512() ? &X86::FR64XRegClass : &X86::FR64RegClass;
1798 else if (RegVT == MVT::f80)
1799 RC = &X86::RFP80RegClass;
1800 else if (RegVT == MVT::f128)
1801 RC = &X86::VR128RegClass;
1802 else if (RegVT.is512BitVector())
1803 RC = &X86::VR512RegClass;
1804 else if (RegVT.is256BitVector())
1805 RC = Subtarget.hasVLX() ? &X86::VR256XRegClass : &X86::VR256RegClass;
1806 else if (RegVT.is128BitVector())
1807 RC = Subtarget.hasVLX() ? &X86::VR128XRegClass : &X86::VR128RegClass;
1808 else if (RegVT == MVT::x86mmx)
1809 RC = &X86::VR64RegClass;
1810 else if (RegVT == MVT::v1i1)
1811 RC = &X86::VK1RegClass;
1812 else if (RegVT == MVT::v8i1)
1813 RC = &X86::VK8RegClass;
1814 else if (RegVT == MVT::v16i1)
1815 RC = &X86::VK16RegClass;
1816 else if (RegVT == MVT::v32i1)
1817 RC = &X86::VK32RegClass;
1818 else if (RegVT == MVT::v64i1)
1819 RC = &X86::VK64RegClass;
1820 else
1821 llvm_unreachable("Unknown argument type!");
1822
1823 Register Reg = MF.addLiveIn(VA.getLocReg(), RC);
1824 ArgValue = DAG.getCopyFromReg(Chain, dl, Reg, RegVT);
1825 }
1826
1827 // If this is an 8 or 16-bit value, it is really passed promoted to 32
1828 // bits. Insert an assert[sz]ext to capture this, then truncate to the
1829 // right size.
1830 if (VA.getLocInfo() == CCValAssign::SExt)
1831 ArgValue = DAG.getNode(ISD::AssertSext, dl, RegVT, ArgValue,
1832 DAG.getValueType(VA.getValVT()));
1833 else if (VA.getLocInfo() == CCValAssign::ZExt)
1834 ArgValue = DAG.getNode(ISD::AssertZext, dl, RegVT, ArgValue,
1835 DAG.getValueType(VA.getValVT()));
1836 else if (VA.getLocInfo() == CCValAssign::BCvt)
1837 ArgValue = DAG.getBitcast(VA.getValVT(), ArgValue);
1838
1839 if (VA.isExtInLoc()) {
1840 // Handle MMX values passed in XMM regs.
1841 if (RegVT.isVector() && VA.getValVT().getScalarType() != MVT::i1)
1842 ArgValue = DAG.getNode(X86ISD::MOVDQ2Q, dl, VA.getValVT(), ArgValue);
1843 else if (VA.getValVT().isVector() &&
1844 VA.getValVT().getScalarType() == MVT::i1 &&
1845 ((VA.getLocVT() == MVT::i64) || (VA.getLocVT() == MVT::i32) ||
1846 (VA.getLocVT() == MVT::i16) || (VA.getLocVT() == MVT::i8))) {
1847 // Promoting a mask type (v*i1) into a register of type i64/i32/i16/i8
1848 ArgValue = lowerRegToMasks(ArgValue, VA.getValVT(), RegVT, dl, DAG);
1849 } else
1850 ArgValue = DAG.getNode(ISD::TRUNCATE, dl, VA.getValVT(), ArgValue);
1851 }
1852 } else {
1853 assert(VA.isMemLoc());
1854 ArgValue =
1855 LowerMemArgument(Chain, CallConv, Ins, dl, DAG, VA, MFI, InsIndex);
1856 }
1857
1858 // If value is passed via pointer - do a load.
1859 if (VA.getLocInfo() == CCValAssign::Indirect &&
1860 !(Ins[I].Flags.isByVal() && VA.isRegLoc())) {
1861 ArgValue =
1862 DAG.getLoad(VA.getValVT(), dl, Chain, ArgValue, MachinePointerInfo());
1863 }
1864
1865 InVals.push_back(ArgValue);
1866 }
1867
1868 for (unsigned I = 0, E = Ins.size(); I != E; ++I) {
1869 if (Ins[I].Flags.isSwiftAsync()) {
1870 auto X86FI = MF.getInfo<X86MachineFunctionInfo>();
1871 if (X86::isExtendedSwiftAsyncFrameSupported(Subtarget, MF))
1872 X86FI->setHasSwiftAsyncContext(true);
1873 else {
1874 int PtrSize = Subtarget.is64Bit() ? 8 : 4;
1875 int FI =
1876 MF.getFrameInfo().CreateStackObject(PtrSize, Align(PtrSize), false);
1877 X86FI->setSwiftAsyncContextFrameIdx(FI);
1878 SDValue St = DAG.getStore(
1879 DAG.getEntryNode(), dl, InVals[I],
1880 DAG.getFrameIndex(FI, PtrSize == 8 ? MVT::i64 : MVT::i32),
1882 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, St, Chain);
1883 }
1884 }
1885
1886 // Swift calling convention does not require we copy the sret argument
1887 // into %rax/%eax for the return. We don't set SRetReturnReg for Swift.
1888 if (CallConv == CallingConv::Swift || CallConv == CallingConv::SwiftTail)
1889 continue;
1890
1891 // All x86 ABIs require that for returning structs by value we copy the
1892 // sret argument into %rax/%eax (depending on ABI) for the return. Save
1893 // the argument into a virtual register so that we can access it from the
1894 // return points.
1895 if (Ins[I].Flags.isSRet()) {
1896 assert(!FuncInfo->getSRetReturnReg() &&
1897 "SRet return has already been set");
1898 MVT PtrTy = getPointerTy(DAG.getDataLayout());
1899 Register Reg =
1901 FuncInfo->setSRetReturnReg(Reg);
1902 SDValue Copy = DAG.getCopyToReg(DAG.getEntryNode(), dl, Reg, InVals[I]);
1903 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, Copy, Chain);
1904 break;
1905 }
1906 }
1907
1908 unsigned StackSize = CCInfo.getStackSize();
1909 // Align stack specially for tail calls.
1910 if (shouldGuaranteeTCO(CallConv,
1912 StackSize = GetAlignedArgumentStackSize(StackSize, DAG);
1913
1914 if (IsVarArg)
1915 VarArgsLoweringHelper(FuncInfo, dl, DAG, Subtarget, CallConv, CCInfo)
1916 .lowerVarArgsParameters(Chain, StackSize);
1917
1918 // Some CCs need callee pop.
1919 if (X86::isCalleePop(CallConv, Is64Bit, IsVarArg,
1921 FuncInfo->setBytesToPopOnReturn(StackSize); // Callee pops everything.
1922 } else if (CallConv == CallingConv::X86_INTR && Ins.size() == 2) {
1923 // X86 interrupts must pop the error code (and the alignment padding) if
1924 // present.
1925 FuncInfo->setBytesToPopOnReturn(Is64Bit ? 16 : 4);
1926 } else {
1927 FuncInfo->setBytesToPopOnReturn(0); // Callee pops nothing.
1928 // If this is an sret function, the return should pop the hidden pointer.
1929 if (hasCalleePopSRet(Ins, ArgLocs, Subtarget))
1930 FuncInfo->setBytesToPopOnReturn(4);
1931 }
1932
1933 if (!Is64Bit) {
1934 // RegSaveFrameIndex is X86-64 only.
1935 FuncInfo->setRegSaveFrameIndex(0xAAAAAAA);
1936 }
1937
1938 FuncInfo->setArgumentStackSize(StackSize);
1939
1940 if (WinEHFuncInfo *EHInfo = MF.getWinEHFuncInfo()) {
1941 EHPersonality Personality = classifyEHPersonality(F.getPersonalityFn());
1942 if (Personality == EHPersonality::CoreCLR) {
1943 assert(Is64Bit);
1944 // TODO: Add a mechanism to frame lowering that will allow us to indicate
1945 // that we'd prefer this slot be allocated towards the bottom of the frame
1946 // (i.e. near the stack pointer after allocating the frame). Every
1947 // funclet needs a copy of this slot in its (mostly empty) frame, and the
1948 // offset from the bottom of this and each funclet's frame must be the
1949 // same, so the size of funclets' (mostly empty) frames is dictated by
1950 // how far this slot is from the bottom (since they allocate just enough
1951 // space to accommodate holding this slot at the correct offset).
1952 int PSPSymFI = MFI.CreateStackObject(8, Align(8), /*isSpillSlot=*/false);
1953 EHInfo->PSPSymFrameIdx = PSPSymFI;
1954 }
1955 }
1956
1957 if (shouldDisableArgRegFromCSR(CallConv) ||
1958 F.hasFnAttribute("no_caller_saved_registers")) {
1959 MachineRegisterInfo &MRI = MF.getRegInfo();
1960 for (std::pair<MCRegister, Register> Pair : MRI.liveins())
1961 MRI.disableCalleeSavedRegister(Pair.first);
1962 }
1963
1964 if (CallingConv::PreserveNone == CallConv)
1965 for (const ISD::InputArg &In : Ins) {
1966 if (In.Flags.isSwiftSelf() || In.Flags.isSwiftAsync() ||
1967 In.Flags.isSwiftError()) {
1968 errorUnsupported(DAG, dl,
1969 "Swift attributes can't be used with preserve_none");
1970 break;
1971 }
1972 }
1973
1974 return Chain;
1975}
1976
1977SDValue X86TargetLowering::LowerMemOpCallTo(SDValue Chain, SDValue StackPtr,
1978 SDValue Arg, const SDLoc &dl,
1979 SelectionDAG &DAG,
1980 const CCValAssign &VA,
1981 ISD::ArgFlagsTy Flags,
1982 bool isByVal) const {
1983 unsigned LocMemOffset = VA.getLocMemOffset();
1984 SDValue PtrOff = DAG.getIntPtrConstant(LocMemOffset, dl);
1985 PtrOff = DAG.getNode(ISD::ADD, dl, getPointerTy(DAG.getDataLayout()),
1986 StackPtr, PtrOff);
1987 if (isByVal)
1988 return CreateCopyOfByValArgument(Arg, PtrOff, Chain, Flags, DAG, dl);
1989
1990 MaybeAlign Alignment;
1991 if (Subtarget.isTargetWindowsMSVC() && !Subtarget.is64Bit() &&
1992 Arg.getSimpleValueType() != MVT::f80)
1993 Alignment = MaybeAlign(4);
1994 return DAG.getStore(
1995 Chain, dl, Arg, PtrOff,
1997 Alignment);
1998}
1999
2000/// Emit a load of return address if tail call
2001/// optimization is performed and it is required.
2002SDValue X86TargetLowering::EmitTailCallLoadRetAddr(
2003 SelectionDAG &DAG, SDValue &OutRetAddr, SDValue Chain, bool IsTailCall,
2004 bool Is64Bit, int FPDiff, const SDLoc &dl) const {
2005 // Adjust the Return address stack slot.
2006 EVT VT = getPointerTy(DAG.getDataLayout());
2007 OutRetAddr = getReturnAddressFrameIndex(DAG);
2008
2009 // Load the "old" Return address.
2010 OutRetAddr = DAG.getLoad(VT, dl, Chain, OutRetAddr, MachinePointerInfo());
2011 return SDValue(OutRetAddr.getNode(), 1);
2012}
2013
2014/// Emit a store of the return address if tail call
2015/// optimization is performed and it is required (FPDiff!=0).
2017 SDValue Chain, SDValue RetAddrFrIdx,
2018 EVT PtrVT, unsigned SlotSize,
2019 int FPDiff, const SDLoc &dl) {
2020 // Store the return address to the appropriate stack slot.
2021 if (!FPDiff) return Chain;
2022 // Calculate the new stack slot for the return address.
2023 int NewReturnAddrFI =
2024 MF.getFrameInfo().CreateFixedObject(SlotSize, (int64_t)FPDiff - SlotSize,
2025 false);
2026 SDValue NewRetAddrFrIdx = DAG.getFrameIndex(NewReturnAddrFI, PtrVT);
2027 Chain = DAG.getStore(Chain, dl, RetAddrFrIdx, NewRetAddrFrIdx,
2029 DAG.getMachineFunction(), NewReturnAddrFI));
2030 return Chain;
2031}
2032
2033/// Returns a vector_shuffle mask for an movs{s|d}, movd
2034/// operation of specified width.
2035SDValue X86TargetLowering::getMOVL(SelectionDAG &DAG, const SDLoc &dl, MVT VT,
2036 SDValue V1, SDValue V2) const {
2037 unsigned NumElems = VT.getVectorNumElements();
2038 SmallVector<int, 8> Mask;
2039 Mask.push_back(NumElems);
2040 for (unsigned i = 1; i != NumElems; ++i)
2041 Mask.push_back(i);
2042 return DAG.getVectorShuffle(VT, dl, V1, V2, Mask);
2043}
2044
2045// Returns the type of copying which is required to set up a byval argument to
2046// a tail-called function. This isn't needed for non-tail calls, because they
2047// always need the equivalent of CopyOnce, but tail-calls sometimes need two to
2048// avoid clobbering another argument (CopyViaTemp), and sometimes can be
2049// optimised to zero copies when forwarding an argument from the caller's
2050// caller (NoCopy).
2051X86TargetLowering::ByValCopyKind X86TargetLowering::ByValNeedsCopyForTailCall(
2052 SelectionDAG &DAG, SDValue Src, SDValue Dst, ISD::ArgFlagsTy Flags) const {
2053 MachineFrameInfo &MFI = DAG.getMachineFunction().getFrameInfo();
2054
2055 // Globals are always safe to copy from.
2057 return CopyOnce;
2058
2059 // Can only analyse frame index nodes, conservatively assume we need a
2060 // temporary.
2061 auto *SrcFrameIdxNode = dyn_cast<FrameIndexSDNode>(Src);
2062 auto *DstFrameIdxNode = dyn_cast<FrameIndexSDNode>(Dst);
2063 if (!SrcFrameIdxNode || !DstFrameIdxNode)
2064 return CopyViaTemp;
2065
2066 int SrcFI = SrcFrameIdxNode->getIndex();
2067 int DstFI = DstFrameIdxNode->getIndex();
2068 assert(MFI.isFixedObjectIndex(DstFI) &&
2069 "byval passed in non-fixed stack slot");
2070
2071 int64_t SrcOffset = MFI.getObjectOffset(SrcFI);
2072 int64_t DstOffset = MFI.getObjectOffset(DstFI);
2073
2074 // If the source is in the local frame, then the copy to the argument
2075 // memory is always valid.
2076 bool FixedSrc = MFI.isFixedObjectIndex(SrcFI);
2077 if (!FixedSrc || (FixedSrc && SrcOffset < 0))
2078 return CopyOnce;
2079
2080 // If the value is already in the correct location, then no copying is
2081 // needed. If not, then we need to copy via a temporary.
2082 if (SrcOffset == DstOffset)
2083 return NoCopy;
2084 else
2085 return CopyViaTemp;
2086}
2087
2088SDValue
2089X86TargetLowering::LowerCall(TargetLowering::CallLoweringInfo &CLI,
2090 SmallVectorImpl<SDValue> &InVals) const {
2091 SelectionDAG &DAG = CLI.DAG;
2092 SDLoc &dl = CLI.DL;
2093 SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
2094 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2095 SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
2096 SDValue Chain = CLI.Chain;
2097 SDValue Callee = CLI.Callee;
2098 CallingConv::ID CallConv = CLI.CallConv;
2099 bool &isTailCall = CLI.IsTailCall;
2100 bool isVarArg = CLI.IsVarArg;
2101 const auto *CB = CLI.CB;
2102
2103 MachineFunction &MF = DAG.getMachineFunction();
2104 bool Is64Bit = Subtarget.is64Bit();
2105 bool IsWin64 = Subtarget.isCallingConvWin64(CallConv);
2106 bool ShouldGuaranteeTCO = shouldGuaranteeTCO(
2107 CallConv, MF.getTarget().Options.GuaranteedTailCallOpt);
2108 X86MachineFunctionInfo *X86Info = MF.getInfo<X86MachineFunctionInfo>();
2109 bool HasNCSR = (CB && isa<CallInst>(CB) &&
2110 CB->hasFnAttr("no_caller_saved_registers"));
2111 bool IsIndirectCall = (CB && isa<CallInst>(CB) && CB->isIndirectCall());
2112 bool IsCFICall = IsIndirectCall && CLI.CFIType;
2113 const Module *M = MF.getFunction().getParent();
2114
2115 // If the indirect call target has the nocf_check attribute, the call needs
2116 // the NOTRACK prefix. For simplicity just disable tail calls as there are
2117 // so many variants.
2118 // FIXME: This will cause backend errors if the user forces the issue.
2119 bool IsNoTrackIndirectCall = IsIndirectCall && CB->doesNoCfCheck() &&
2120 M->getModuleFlag("cf-protection-branch");
2121 if (IsNoTrackIndirectCall)
2122 isTailCall = false;
2123
2124 MachineFunction::CallSiteInfo CSInfo;
2125 if (CallConv == CallingConv::X86_INTR)
2126 report_fatal_error("X86 interrupts may not be called directly");
2127
2128 // Set type id for call site info.
2129 setTypeIdForCallsiteInfo(CB, MF, CSInfo);
2130
2131 if (IsIndirectCall && !IsWin64 &&
2132 M->getModuleFlag("import-call-optimization"))
2133 errorUnsupported(DAG, dl,
2134 "Indirect calls must have a normal calling convention if "
2135 "Import Call Optimization is enabled");
2136
2137 // Analyze operands of the call, assigning locations to each operand.
2139 CCState CCInfo(CallConv, isVarArg, MF, ArgLocs, *DAG.getContext());
2140
2141 // Allocate shadow area for Win64.
2142 if (IsWin64)
2143 CCInfo.AllocateStack(32, Align(8));
2144
2145 CCInfo.AnalyzeArguments(Outs, CC_X86);
2146
2147 // In vectorcall calling convention a second pass is required for the HVA
2148 // types.
2149 if (CallingConv::X86_VectorCall == CallConv) {
2150 CCInfo.AnalyzeArgumentsSecondPass(Outs, CC_X86);
2151 }
2152
2153 // We cannot guarantee TCO for mismatched calling conventions.
2154 if (isTailCall && ShouldGuaranteeTCO) {
2155 CallingConv::ID CallerCC = MF.getFunction().getCallingConv();
2156 isTailCall = (CallConv == CallerCC);
2157 }
2158
2159 // Check if this tail call is a "sibling" call, which is loosely defined to
2160 // be a tail call that doesn't require heroics like moving the return
2161 // address or swapping byval arguments. We treat some musttail calls as
2162 // sibling calls to avoid unnecessary argument copies.
2163 bool IsMustTail = CLI.CB && CLI.CB->isMustTailCall();
2164 bool IsSibcall = false;
2165 if (isTailCall) {
2166 IsSibcall = isEligibleForSiblingCallOpt(CLI, CCInfo, ArgLocs);
2167 isTailCall = IsSibcall || IsMustTail || ShouldGuaranteeTCO;
2168 }
2169
2170 if (isTailCall)
2171 ++NumTailCalls;
2172
2173 if (IsMustTail && !isTailCall)
2174 report_fatal_error("failed to perform tail call elimination on a call "
2175 "site marked musttail");
2176
2177 assert(!(isVarArg && canGuaranteeTCO(CallConv)) &&
2178 "Var args not supported with calling convention fastcc, ghc or hipe");
2179
2180 // Get a count of how many bytes are to be pushed on the stack.
2181 unsigned NumBytes = CCInfo.getAlignedCallFrameSize();
2182 if (IsSibcall)
2183 // This is a sibcall. The memory operands are available in caller's
2184 // own caller's stack.
2185 NumBytes = 0;
2186 else if (ShouldGuaranteeTCO && canGuaranteeTCO(CallConv))
2187 NumBytes = GetAlignedArgumentStackSize(NumBytes, DAG);
2188
2189 // A sibcall is ABI-compatible and does not need to adjust the stack pointer.
2190 int FPDiff = 0;
2191 if (isTailCall && ShouldGuaranteeTCO && !IsSibcall) {
2192 // Lower arguments at fp - stackoffset + fpdiff.
2193 unsigned NumBytesCallerPushed = X86Info->getBytesToPopOnReturn();
2194
2195 FPDiff = NumBytesCallerPushed - NumBytes;
2196
2197 // Set the delta of movement of the returnaddr stackslot.
2198 // But only set if delta is greater than previous delta.
2199 if (FPDiff < X86Info->getTCReturnAddrDelta())
2200 X86Info->setTCReturnAddrDelta(FPDiff);
2201 }
2202
2203 unsigned NumBytesToPush = NumBytes;
2204 unsigned NumBytesToPop = NumBytes;
2205
2207 const X86RegisterInfo *RegInfo = Subtarget.getRegisterInfo();
2208
2209 // If we are doing a tail-call, any byval arguments will be written to stack
2210 // space which was used for incoming arguments. If any the values being used
2211 // are incoming byval arguments to this function, then they might be
2212 // overwritten by the stores of the outgoing arguments. To avoid this, we
2213 // need to make a temporary copy of them in local stack space, then copy back
2214 // to the argument area.
2215 // FIXME: There's potential to improve the code by using virtual registers for
2216 // temporary storage, and letting the register allocator spill if needed.
2217 SmallVector<SDValue, 8> ByValTemporaries;
2218 SDValue ByValTempChain;
2219 if (isTailCall) {
2220 // Use null SDValue to mean "no temporary recorded for this arg index".
2221 ByValTemporaries.assign(OutVals.size(), SDValue());
2222
2223 SmallVector<SDValue, 8> ByValCopyChains;
2224 for (const CCValAssign &VA : ArgLocs) {
2225 unsigned ArgIdx = VA.getValNo();
2226 SDValue Src = OutVals[ArgIdx];
2227 ISD::ArgFlagsTy Flags = Outs[ArgIdx].Flags;
2228
2229 if (!Flags.isByVal())
2230 continue;
2231
2232 auto PtrVT = getPointerTy(DAG.getDataLayout());
2233
2234 if (!StackPtr.getNode())
2235 StackPtr =
2236 DAG.getCopyFromReg(Chain, dl, RegInfo->getStackRegister(), PtrVT);
2237
2238 // Destination: where this byval should live in the callee’s frame
2239 // after the tail call.
2240 int64_t Offset = VA.getLocMemOffset() + FPDiff;
2241 uint64_t Size = VA.getLocVT().getFixedSizeInBits() / 8;
2243 /*IsImmutable=*/true);
2244 SDValue Dst = DAG.getFrameIndex(FI, PtrVT);
2245
2246 ByValCopyKind Copy = ByValNeedsCopyForTailCall(DAG, Src, Dst, Flags);
2247
2248 if (Copy == NoCopy) {
2249 // If the argument is already at the correct offset on the stack
2250 // (because we are forwarding a byval argument from our caller), we
2251 // don't need any copying.
2252 continue;
2253 } else if (Copy == CopyOnce) {
2254 // If the argument is in our local stack frame, no other argument
2255 // preparation can clobber it, so we can copy it to the final location
2256 // later.
2257 ByValTemporaries[ArgIdx] = Src;
2258 } else {
2259 assert(Copy == CopyViaTemp && "unexpected enum value");
2260 // If we might be copying this argument from the outgoing argument
2261 // stack area, we need to copy via a temporary in the local stack
2262 // frame.
2263 MachineFrameInfo &MFI = MF.getFrameInfo();
2264 int TempFrameIdx = MFI.CreateStackObject(Flags.getByValSize(),
2265 Flags.getNonZeroByValAlign(),
2266 /*isSS=*/false);
2267 SDValue Temp =
2268 DAG.getFrameIndex(TempFrameIdx, getPointerTy(DAG.getDataLayout()));
2269
2270 SDValue CopyChain =
2271 CreateCopyOfByValArgument(Src, Temp, Chain, Flags, DAG, dl);
2272 ByValCopyChains.push_back(CopyChain);
2273 ByValTemporaries[ArgIdx] = Temp;
2274 }
2275 }
2276 if (!ByValCopyChains.empty())
2277 ByValTempChain =
2278 DAG.getNode(ISD::TokenFactor, dl, MVT::Other, ByValCopyChains);
2279 }
2280
2281 // If we have an inalloca argument, all stack space has already been allocated
2282 // for us and be right at the top of the stack. We don't support multiple
2283 // arguments passed in memory when using inalloca.
2284 if (!Outs.empty() && Outs.back().Flags.isInAlloca()) {
2285 NumBytesToPush = 0;
2286 if (!ArgLocs.back().isMemLoc())
2287 report_fatal_error("cannot use inalloca attribute on a register "
2288 "parameter");
2289 if (ArgLocs.back().getLocMemOffset() != 0)
2290 report_fatal_error("any parameter with the inalloca attribute must be "
2291 "the only memory argument");
2292 } else if (CLI.IsPreallocated) {
2293 assert(ArgLocs.back().isMemLoc() &&
2294 "cannot use preallocated attribute on a register "
2295 "parameter");
2296 SmallVector<size_t, 4> PreallocatedOffsets;
2297 for (size_t i = 0; i < CLI.OutVals.size(); ++i) {
2298 if (CLI.CB->paramHasAttr(i, Attribute::Preallocated)) {
2299 PreallocatedOffsets.push_back(ArgLocs[i].getLocMemOffset());
2300 }
2301 }
2302 auto *MFI = DAG.getMachineFunction().getInfo<X86MachineFunctionInfo>();
2303 size_t PreallocatedId = MFI->getPreallocatedIdForCallSite(CLI.CB);
2304 MFI->setPreallocatedStackSize(PreallocatedId, NumBytes);
2305 MFI->setPreallocatedArgOffsets(PreallocatedId, PreallocatedOffsets);
2306 NumBytesToPush = 0;
2307 }
2308
2309 if (!IsSibcall && !IsMustTail)
2310 Chain = DAG.getCALLSEQ_START(Chain, NumBytesToPush,
2311 NumBytes - NumBytesToPush, dl);
2312
2313 SDValue RetAddrFrIdx;
2314 // Load return address for tail calls.
2315 if (isTailCall && FPDiff)
2316 Chain = EmitTailCallLoadRetAddr(DAG, RetAddrFrIdx, Chain, isTailCall,
2317 Is64Bit, FPDiff, dl);
2318
2320 SmallVector<SDValue, 8> MemOpChains;
2321
2322 // The next loop assumes that the locations are in the same order of the
2323 // input arguments.
2324 assert(isSortedByValueNo(ArgLocs) &&
2325 "Argument Location list must be sorted before lowering");
2326
2327 // Walk the register/memloc assignments, inserting copies/loads. In the case
2328 // of tail call optimization arguments are handle later.
2329 for (unsigned I = 0, OutIndex = 0, E = ArgLocs.size(); I != E;
2330 ++I, ++OutIndex) {
2331 assert(OutIndex < Outs.size() && "Invalid Out index");
2332 // Skip inalloca/preallocated arguments, they have already been written.
2333 ISD::ArgFlagsTy Flags = Outs[OutIndex].Flags;
2334 if (Flags.isInAlloca() || Flags.isPreallocated())
2335 continue;
2336
2337 CCValAssign &VA = ArgLocs[I];
2338 EVT RegVT = VA.getLocVT();
2339 SDValue Arg = OutVals[OutIndex];
2340 bool isByVal = Flags.isByVal();
2341
2342 // Promote the value if needed.
2343 switch (VA.getLocInfo()) {
2344 default: llvm_unreachable("Unknown loc info!");
2345 case CCValAssign::Full: break;
2346 case CCValAssign::SExt:
2347 Arg = DAG.getNode(ISD::SIGN_EXTEND, dl, RegVT, Arg);
2348 break;
2349 case CCValAssign::ZExt:
2350 Arg = DAG.getNode(ISD::ZERO_EXTEND, dl, RegVT, Arg);
2351 break;
2352 case CCValAssign::AExt:
2353 if (Arg.getValueType().isVector() &&
2354 Arg.getValueType().getVectorElementType() == MVT::i1)
2355 Arg = lowerMasksToReg(Arg, RegVT, dl, DAG);
2356 else if (RegVT.is128BitVector()) {
2357 // Special case: passing MMX values in XMM registers.
2358 Arg = DAG.getBitcast(MVT::i64, Arg);
2359 Arg = DAG.getNode(ISD::SCALAR_TO_VECTOR, dl, MVT::v2i64, Arg);
2360 Arg = getMOVL(DAG, dl, MVT::v2i64, DAG.getUNDEF(MVT::v2i64), Arg);
2361 } else
2362 Arg = DAG.getNode(ISD::ANY_EXTEND, dl, RegVT, Arg);
2363 break;
2364 case CCValAssign::BCvt:
2365 Arg = DAG.getBitcast(RegVT, Arg);
2366 break;
2367 case CCValAssign::Indirect: {
2368 if (isByVal) {
2369 // Memcpy the argument to a temporary stack slot to prevent
2370 // the caller from seeing any modifications the callee may make
2371 // as guaranteed by the `byval` attribute.
2372 int FrameIdx = MF.getFrameInfo().CreateStackObject(
2373 Flags.getByValSize(),
2374 std::max(Align(16), Flags.getNonZeroByValAlign()), false);
2375 SDValue StackSlot =
2376 DAG.getFrameIndex(FrameIdx, getPointerTy(DAG.getDataLayout()));
2377 Chain =
2378 CreateCopyOfByValArgument(Arg, StackSlot, Chain, Flags, DAG, dl);
2379 // From now on treat this as a regular pointer
2380 Arg = StackSlot;
2381 isByVal = false;
2382 } else {
2383 // Store the argument.
2384 SDValue SpillSlot = DAG.CreateStackTemporary(VA.getValVT());
2385 int FI = cast<FrameIndexSDNode>(SpillSlot)->getIndex();
2386 Chain = DAG.getStore(
2387 Chain, dl, Arg, SpillSlot,
2389 Arg = SpillSlot;
2390 }
2391 break;
2392 }
2393 }
2394
2395 if (VA.needsCustom()) {
2396 assert(VA.getValVT() == MVT::v64i1 &&
2397 "Currently the only custom case is when we split v64i1 to 2 regs");
2398 // Split v64i1 value into two registers
2399 Passv64i1ArgInRegs(dl, DAG, Arg, RegsToPass, VA, ArgLocs[++I], Subtarget);
2400 } else if (VA.isRegLoc()) {
2401 RegsToPass.push_back(std::make_pair(VA.getLocReg(), Arg));
2402 const TargetOptions &Options = DAG.getTarget().Options;
2403 if (Options.EmitCallSiteInfo)
2404 CSInfo.ArgRegPairs.emplace_back(VA.getLocReg(), I);
2405 if (isVarArg && IsWin64) {
2406 // Win64 ABI requires argument XMM reg to be copied to the corresponding
2407 // shadow reg if callee is a varargs function.
2408 Register ShadowReg;
2409 switch (VA.getLocReg()) {
2410 case X86::XMM0: ShadowReg = X86::RCX; break;
2411 case X86::XMM1: ShadowReg = X86::RDX; break;
2412 case X86::XMM2: ShadowReg = X86::R8; break;
2413 case X86::XMM3: ShadowReg = X86::R9; break;
2414 }
2415 if (ShadowReg)
2416 RegsToPass.push_back(std::make_pair(ShadowReg, Arg));
2417 }
2418 } else if (!IsSibcall && (!isTailCall || (isByVal && !IsMustTail))) {
2419 assert(VA.isMemLoc());
2420 if (!StackPtr.getNode())
2421 StackPtr = DAG.getCopyFromReg(Chain, dl, RegInfo->getStackRegister(),
2423 MemOpChains.push_back(LowerMemOpCallTo(Chain, StackPtr, Arg,
2424 dl, DAG, VA, Flags, isByVal));
2425 }
2426 }
2427
2428 if (!MemOpChains.empty())
2429 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOpChains);
2430
2431 if (Subtarget.isPICStyleGOT()) {
2432 // ELF / PIC requires GOT in the EBX register before function calls via PLT
2433 // GOT pointer.
2434 if (!isTailCall) {
2435 // Only PLT calls (GlobalAddress or ExternalSymbol) require the GOT in
2436 // EBX. Indirect calls through a register or an absolute address do not
2437 // go through the PLT and do not need EBX to hold the GOT base.
2438 if ((Callee->getOpcode() == ISD::GlobalAddress ||
2439 Callee->getOpcode() == ISD::ExternalSymbol))
2440 RegsToPass.push_back(std::make_pair(
2441 Register(X86::EBX), DAG.getNode(X86ISD::GlobalBaseReg, SDLoc(),
2442 getPointerTy(DAG.getDataLayout()))));
2443 } else {
2444 // If we are tail calling and generating PIC/GOT style code load the
2445 // address of the callee into ECX. The value in ecx is used as target of
2446 // the tail jump. This is done to circumvent the ebx/callee-saved problem
2447 // for tail calls on PIC/GOT architectures. Normally we would just put the
2448 // address of GOT into ebx and then call target@PLT. But for tail calls
2449 // ebx would be restored (since ebx is callee saved) before jumping to the
2450 // target@PLT.
2451
2452 // Note: The actual moving to ECX is done further down.
2453 GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee);
2454 if (G && !G->getGlobal()->hasLocalLinkage() &&
2455 G->getGlobal()->hasDefaultVisibility())
2456 Callee = LowerGlobalAddress(Callee, DAG);
2457 else if (isa<ExternalSymbolSDNode>(Callee))
2458 Callee = LowerExternalSymbol(Callee, DAG);
2459 }
2460 }
2461
2462 if (Is64Bit && isVarArg && !IsWin64 && !IsMustTail &&
2463 (Subtarget.hasSSE1() || !M->getModuleFlag("SkipRaxSetup"))) {
2464 // From AMD64 ABI document:
2465 // For calls that may call functions that use varargs or stdargs
2466 // (prototype-less calls or calls to functions containing ellipsis (...) in
2467 // the declaration) %al is used as hidden argument to specify the number
2468 // of SSE registers used. The contents of %al do not need to match exactly
2469 // the number of registers, but must be an ubound on the number of SSE
2470 // registers used and is in the range 0 - 8 inclusive.
2471
2472 // Count the number of XMM registers allocated.
2473 static const MCPhysReg XMMArgRegs[] = {
2474 X86::XMM0, X86::XMM1, X86::XMM2, X86::XMM3,
2475 X86::XMM4, X86::XMM5, X86::XMM6, X86::XMM7
2476 };
2477 unsigned NumXMMRegs = CCInfo.getFirstUnallocated(XMMArgRegs);
2478 assert((Subtarget.hasSSE1() || !NumXMMRegs)
2479 && "SSE registers cannot be used when SSE is disabled");
2480 RegsToPass.push_back(std::make_pair(Register(X86::AL),
2481 DAG.getConstant(NumXMMRegs, dl,
2482 MVT::i8)));
2483 }
2484
2485 if (isVarArg && IsMustTail) {
2486 const auto &Forwards = X86Info->getForwardedMustTailRegParms();
2487 for (const auto &F : Forwards) {
2488 SDValue Val = DAG.getCopyFromReg(Chain, dl, F.VReg, F.VT);
2489 RegsToPass.push_back(std::make_pair(F.PReg, Val));
2490 }
2491 }
2492
2493 // For tail calls lower the arguments to the 'real' stack slots. Sibcalls
2494 // don't need this because the eligibility check rejects calls that require
2495 // shuffling arguments passed in memory.
2496 if (isTailCall && !IsSibcall) {
2497 // Force all the incoming stack arguments to be loaded from the stack
2498 // before any new outgoing arguments or the return address are stored to the
2499 // stack, because the outgoing stack slots may alias the incoming argument
2500 // stack slots, and the alias isn't otherwise explicit. This is slightly
2501 // more conservative than necessary, because it means that each store
2502 // effectively depends on every argument instead of just those arguments it
2503 // would clobber.
2504 Chain = DAG.getStackArgumentTokenFactor(Chain);
2505
2506 if (ByValTempChain)
2507 Chain =
2508 DAG.getNode(ISD::TokenFactor, dl, MVT::Other, Chain, ByValTempChain);
2509
2510 SmallVector<SDValue, 8> MemOpChains2;
2511 SDValue FIN;
2512 int FI = 0;
2513 for (unsigned I = 0, OutsIndex = 0, E = ArgLocs.size(); I != E;
2514 ++I, ++OutsIndex) {
2515 CCValAssign &VA = ArgLocs[I];
2516
2517 if (VA.isRegLoc()) {
2518 if (VA.needsCustom()) {
2519 assert((CallConv == CallingConv::X86_RegCall) &&
2520 "Expecting custom case only in regcall calling convention");
2521 // This means that we are in special case where one argument was
2522 // passed through two register locations - Skip the next location
2523 ++I;
2524 }
2525
2526 continue;
2527 }
2528
2529 assert(VA.isMemLoc());
2530 SDValue Arg = OutVals[OutsIndex];
2531 ISD::ArgFlagsTy Flags = Outs[OutsIndex].Flags;
2532 // Skip inalloca/preallocated arguments. They don't require any work.
2533 if (Flags.isInAlloca() || Flags.isPreallocated())
2534 continue;
2535 // Create frame index.
2536 int32_t Offset = VA.getLocMemOffset()+FPDiff;
2537 uint32_t OpSize = (VA.getLocVT().getSizeInBits()+7)/8;
2538 FI = MF.getFrameInfo().CreateFixedObject(OpSize, Offset, true);
2539 FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
2540
2541 if (Flags.isByVal()) {
2542 if (SDValue ByValSrc = ByValTemporaries[OutsIndex]) {
2543 auto PtrVT = getPointerTy(DAG.getDataLayout());
2544 SDValue DstAddr = DAG.getFrameIndex(FI, PtrVT);
2545
2547 ByValSrc, DstAddr, Chain, Flags, DAG, dl));
2548 }
2549 } else {
2550 // Store relative to framepointer.
2551 MemOpChains2.push_back(DAG.getStore(
2552 Chain, dl, Arg, FIN,
2554 }
2555 }
2556
2557 if (!MemOpChains2.empty())
2558 Chain = DAG.getNode(ISD::TokenFactor, dl, MVT::Other, MemOpChains2);
2559
2560 // Store the return address to the appropriate stack slot.
2561 Chain = EmitTailCallStoreRetAddr(DAG, MF, Chain, RetAddrFrIdx,
2563 RegInfo->getSlotSize(), FPDiff, dl);
2564 }
2565
2566 // Build a sequence of copy-to-reg nodes chained together with token chain
2567 // and glue operands which copy the outgoing args into registers.
2568 SDValue InGlue;
2569 for (const auto &[Reg, N] : RegsToPass) {
2570 Chain = DAG.getCopyToReg(Chain, dl, Reg, N, InGlue);
2571 InGlue = Chain.getValue(1);
2572 }
2573
2574 bool IsImpCall = false;
2575 bool IsCFGuardCall = false;
2576 if (DAG.getTarget().getCodeModel() == CodeModel::Large) {
2577 assert(Is64Bit && "Large code model is only legal in 64-bit mode.");
2578 // In the 64-bit large code model, we have to make all calls
2579 // through a register, since the call instruction's 32-bit
2580 // pc-relative offset may not be large enough to hold the whole
2581 // address.
2582 } else if (Callee->getOpcode() == ISD::GlobalAddress ||
2583 Callee->getOpcode() == ISD::ExternalSymbol) {
2584 // Lower direct calls to global addresses and external symbols. Setting
2585 // ForCall to true here has the effect of removing WrapperRIP when possible
2586 // to allow direct calls to be selected without first materializing the
2587 // address into a register.
2588 Callee = LowerGlobalOrExternal(Callee, DAG, /*ForCall=*/true, &IsImpCall);
2589 } else if (Subtarget.isTarget64BitILP32() &&
2590 Callee.getValueType() == MVT::i32) {
2591 // Zero-extend the 32-bit Callee address into a 64-bit according to x32 ABI
2592 Callee = DAG.getNode(ISD::ZERO_EXTEND, dl, MVT::i64, Callee);
2593 } else if (Is64Bit && CB && isCFGuardCall(CB)) {
2594 // We'll use a specific psuedo instruction for tail calls to control flow
2595 // guard functions to guarantee the instruction used for the call. To do
2596 // this we need to unwrap the load now and use the CFG Func GV as the
2597 // callee.
2598 IsCFGuardCall = true;
2599 auto *LoadNode = cast<LoadSDNode>(Callee);
2600 GlobalAddressSDNode *GA =
2601 cast<GlobalAddressSDNode>(unwrapAddress(LoadNode->getBasePtr()));
2603 "CFG Call should be to a guard function");
2604 assert(LoadNode->getOffset()->isUndef() &&
2605 "CFG Function load should not have an offset");
2607 GA->getGlobal(), dl, GA->getValueType(0), 0, X86II::MO_NO_FLAG);
2608 }
2609
2611
2612 if (!IsSibcall && isTailCall && !IsMustTail) {
2613 Chain = DAG.getCALLSEQ_END(Chain, NumBytesToPop, 0, InGlue, dl);
2614 InGlue = Chain.getValue(1);
2615 }
2616
2617 Ops.push_back(Chain);
2618 Ops.push_back(Callee);
2619
2620 if (isTailCall)
2621 Ops.push_back(DAG.getSignedTargetConstant(FPDiff, dl, MVT::i32));
2622
2623 // Add argument registers to the end of the list so that they are known live
2624 // into the call.
2625 for (const auto &[Reg, N] : RegsToPass)
2626 Ops.push_back(DAG.getRegister(Reg, N.getValueType()));
2627
2628 // Add a register mask operand representing the call-preserved registers.
2629 const uint32_t *Mask = [&]() {
2630 auto AdaptedCC = CallConv;
2631 // If HasNCSR is asserted (attribute NoCallerSavedRegisters exists),
2632 // use X86_INTR calling convention because it has the same CSR mask
2633 // (same preserved registers).
2634 if (HasNCSR)
2636 // If NoCalleeSavedRegisters is requested, than use GHC since it happens
2637 // to use the CSR_NoRegs_RegMask.
2638 if (CB && CB->hasFnAttr("no_callee_saved_registers"))
2639 AdaptedCC = (CallingConv::ID)CallingConv::GHC;
2640 return RegInfo->getCallPreservedMask(MF, AdaptedCC);
2641 }();
2642 assert(Mask && "Missing call preserved mask for calling convention");
2643
2644 if (MachineOperand::clobbersPhysReg(Mask, RegInfo->getFramePtr())) {
2645 X86Info->setFPClobberedByCall(true);
2646 if (CLI.CB && isa<InvokeInst>(CLI.CB))
2647 X86Info->setFPClobberedByInvoke(true);
2648 }
2649 if (MachineOperand::clobbersPhysReg(Mask, RegInfo->getBaseRegister())) {
2650 X86Info->setBPClobberedByCall(true);
2651 if (CLI.CB && isa<InvokeInst>(CLI.CB))
2652 X86Info->setBPClobberedByInvoke(true);
2653 }
2654
2655 // If this is an invoke in a 32-bit function using a funclet-based
2656 // personality, assume the function clobbers all registers. If an exception
2657 // is thrown, the runtime will not restore CSRs.
2658 // FIXME: Model this more precisely so that we can register allocate across
2659 // the normal edge and spill and fill across the exceptional edge.
2660 if (!Is64Bit && CLI.CB && isa<InvokeInst>(CLI.CB)) {
2661 const Function &CallerFn = MF.getFunction();
2662 EHPersonality Pers =
2663 CallerFn.hasPersonalityFn()
2666 if (isFuncletEHPersonality(Pers))
2667 Mask = RegInfo->getNoPreservedMask();
2668 }
2669
2670 // Define a new register mask from the existing mask.
2671 uint32_t *RegMask = nullptr;
2672
2673 // In some calling conventions we need to remove the used physical registers
2674 // from the reg mask. Create a new RegMask for such calling conventions.
2675 // RegMask for calling conventions that disable only return registers (e.g.
2676 // preserve_most) will be modified later in LowerCallResult.
2677 bool ShouldDisableArgRegs = shouldDisableArgRegFromCSR(CallConv) || HasNCSR;
2678 if (ShouldDisableArgRegs || shouldDisableRetRegFromCSR(CallConv)) {
2679 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
2680
2681 // Allocate a new Reg Mask and copy Mask.
2682 RegMask = MF.allocateRegMask();
2683 unsigned RegMaskSize = MachineOperand::getRegMaskSize(TRI->getNumRegs());
2684 memcpy(RegMask, Mask, sizeof(RegMask[0]) * RegMaskSize);
2685
2686 // Make sure all sub registers of the argument registers are reset
2687 // in the RegMask.
2688 if (ShouldDisableArgRegs) {
2689 for (auto const &RegPair : RegsToPass)
2690 for (MCPhysReg SubReg : TRI->subregs_inclusive(RegPair.first))
2691 RegMask[SubReg / 32] &= ~(1u << (SubReg % 32));
2692 }
2693
2694 // Create the RegMask Operand according to our updated mask.
2695 Ops.push_back(DAG.getRegisterMask(RegMask));
2696 } else {
2697 // Create the RegMask Operand according to the static mask.
2698 Ops.push_back(DAG.getRegisterMask(Mask));
2699 }
2700
2701 if (InGlue.getNode())
2702 Ops.push_back(InGlue);
2703
2704 if (isTailCall) {
2705 // We used to do:
2706 //// If this is the first return lowered for this function, add the regs
2707 //// to the liveout set for the function.
2708 // This isn't right, although it's probably harmless on x86; liveouts
2709 // should be computed from returns not tail calls. Consider a void
2710 // function making a tail call to a function returning int.
2712 auto Opcode =
2713 IsCFGuardCall ? X86ISD::TC_RETURN_GLOBALADDR : X86ISD::TC_RETURN;
2714 SDValue Ret = DAG.getNode(Opcode, dl, MVT::Other, Ops);
2715
2716 if (IsCFICall)
2717 Ret.getNode()->setCFIType(CLI.CFIType->getZExtValue());
2718
2719 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
2720 DAG.addCallSiteInfo(Ret.getNode(), std::move(CSInfo));
2721 return Ret;
2722 }
2723
2724 // Returns a chain & a glue for retval copy to use.
2725 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
2726 if (IsImpCall) {
2727 Chain = DAG.getNode(X86ISD::IMP_CALL, dl, NodeTys, Ops);
2728 } else if (IsNoTrackIndirectCall) {
2729 Chain = DAG.getNode(X86ISD::NT_CALL, dl, NodeTys, Ops);
2730 } else if (IsCFGuardCall) {
2731 Chain = DAG.getNode(X86ISD::CALL_GLOBALADDR, dl, NodeTys, Ops);
2732 } else if (CLI.CB && objcarc::hasAttachedCallOpBundle(CLI.CB)) {
2733 // Calls with a "clang.arc.attachedcall" bundle are special. They should be
2734 // expanded to the call, directly followed by a special marker sequence and
2735 // a call to a ObjC library function. Use the CALL_RVMARKER to do that.
2736 assert(!isTailCall &&
2737 "tail calls cannot be marked with clang.arc.attachedcall");
2738 assert(Is64Bit && "clang.arc.attachedcall is only supported in 64bit mode");
2739
2740 // Add a target global address for the retainRV/claimRV runtime function
2741 // just before the call target.
2743 auto PtrVT = getPointerTy(DAG.getDataLayout());
2744 auto GA = DAG.getTargetGlobalAddress(ARCFn, dl, PtrVT);
2745 Ops.insert(Ops.begin() + 1, GA);
2746 Chain = DAG.getNode(X86ISD::CALL_RVMARKER, dl, NodeTys, Ops);
2747 } else {
2748 Chain = DAG.getNode(X86ISD::CALL, dl, NodeTys, Ops);
2749 }
2750
2751 if (IsCFICall)
2752 Chain.getNode()->setCFIType(CLI.CFIType->getZExtValue());
2753
2754 InGlue = Chain.getValue(1);
2755 DAG.addNoMergeSiteInfo(Chain.getNode(), CLI.NoMerge);
2756 DAG.addCallSiteInfo(Chain.getNode(), std::move(CSInfo));
2757
2758 // Save heapallocsite metadata.
2759 if (CLI.CB)
2760 if (MDNode *HeapAlloc = CLI.CB->getMetadata("heapallocsite"))
2761 DAG.addHeapAllocSite(Chain.getNode(), HeapAlloc);
2762
2763 // Create the CALLSEQ_END node.
2764 unsigned NumBytesForCalleeToPop = 0; // Callee pops nothing.
2765 if (X86::isCalleePop(CallConv, Is64Bit, isVarArg,
2767 NumBytesForCalleeToPop = NumBytes; // Callee pops everything
2768 } else if (hasCalleePopSRet(Outs, ArgLocs, Subtarget)) {
2769 // If this call passes a struct-return pointer, the callee
2770 // pops that struct pointer.
2771 NumBytesForCalleeToPop = 4;
2772 }
2773
2774 // Returns a glue for retval copy to use.
2775 if (!IsSibcall) {
2776 Chain = DAG.getCALLSEQ_END(Chain, NumBytesToPop, NumBytesForCalleeToPop,
2777 InGlue, dl);
2778 InGlue = Chain.getValue(1);
2779 }
2780
2781 if (CallingConv::PreserveNone == CallConv)
2782 for (const ISD::OutputArg &Out : Outs) {
2783 if (Out.Flags.isSwiftSelf() || Out.Flags.isSwiftAsync() ||
2784 Out.Flags.isSwiftError()) {
2785 errorUnsupported(DAG, dl,
2786 "Swift attributes can't be used with preserve_none");
2787 break;
2788 }
2789 }
2790
2791 // Handle result values, copying them out of physregs into vregs that we
2792 // return.
2793 return LowerCallResult(Chain, InGlue, CallConv, isVarArg, Ins, dl, DAG,
2794 InVals, RegMask);
2795}
2796
2797//===----------------------------------------------------------------------===//
2798// Fast Calling Convention (tail call) implementation
2799//===----------------------------------------------------------------------===//
2800
2801// Like std call, callee cleans arguments, convention except that ECX is
2802// reserved for storing the tail called function address. Only 2 registers are
2803// free for argument passing (inreg). Tail call optimization is performed
2804// provided:
2805// * tailcallopt is enabled
2806// * caller/callee are fastcc
2807// On X86_64 architecture with GOT-style position independent code only local
2808// (within module) calls are supported at the moment.
2809// To keep the stack aligned according to platform abi the function
2810// GetAlignedArgumentStackSize ensures that argument delta is always multiples
2811// of stack alignment. (Dynamic linkers need this - Darwin's dyld for example)
2812// If a tail called function callee has more arguments than the caller the
2813// caller needs to make sure that there is room to move the RETADDR to. This is
2814// achieved by reserving an area the size of the argument delta right after the
2815// original RETADDR, but before the saved framepointer or the spilled registers
2816// e.g. caller(arg1, arg2) calls callee(arg1, arg2,arg3,arg4)
2817// stack layout:
2818// arg1
2819// arg2
2820// RETADDR
2821// [ new RETADDR
2822// move area ]
2823// (possible EBP)
2824// ESI
2825// EDI
2826// local1 ..
2827
2828/// Make the stack size align e.g 16n + 12 aligned for a 16-byte align
2829/// requirement.
2830unsigned
2831X86TargetLowering::GetAlignedArgumentStackSize(const unsigned StackSize,
2832 SelectionDAG &DAG) const {
2833 const Align StackAlignment = Subtarget.getFrameLowering()->getStackAlign();
2834 const uint64_t SlotSize = Subtarget.getRegisterInfo()->getSlotSize();
2835 assert(StackSize % SlotSize == 0 &&
2836 "StackSize must be a multiple of SlotSize");
2837 return alignTo(StackSize + SlotSize, StackAlignment) - SlotSize;
2838}
2839
2840/// Return true if the given stack call argument is already available in the
2841/// same position (relatively) of the caller's incoming argument stack.
2842static
2844 MachineFrameInfo &MFI, const MachineRegisterInfo *MRI,
2845 const X86InstrInfo *TII, const CCValAssign &VA) {
2846 unsigned Bytes = Arg.getValueSizeInBits() / 8;
2847
2848 for (;;) {
2849 // Look through nodes that don't alter the bits of the incoming value.
2850 unsigned Op = Arg.getOpcode();
2851 if (Op == ISD::ZERO_EXTEND || Op == ISD::ANY_EXTEND || Op == ISD::BITCAST ||
2852 Op == ISD::AssertZext) {
2853 Arg = Arg.getOperand(0);
2854 continue;
2855 }
2856 if (Op == ISD::TRUNCATE) {
2857 const SDValue &TruncInput = Arg.getOperand(0);
2858 if (TruncInput.getOpcode() == ISD::AssertZext &&
2859 cast<VTSDNode>(TruncInput.getOperand(1))->getVT() ==
2860 Arg.getValueType()) {
2861 Arg = TruncInput.getOperand(0);
2862 continue;
2863 }
2864 }
2865 break;
2866 }
2867
2868 int FI = INT_MAX;
2869 if (Arg.getOpcode() == ISD::CopyFromReg) {
2870 Register VR = cast<RegisterSDNode>(Arg.getOperand(1))->getReg();
2871 if (!VR.isVirtual())
2872 return false;
2873 MachineInstr *Def = MRI->getVRegDef(VR);
2874 if (!Def)
2875 return false;
2876 if (!Flags.isByVal()) {
2877 if (!TII->isLoadFromStackSlot(*Def, FI))
2878 return false;
2879 } else {
2880 unsigned Opcode = Def->getOpcode();
2881 if ((Opcode == X86::LEA32r || Opcode == X86::LEA64r ||
2882 Opcode == X86::LEA64_32r) &&
2883 Def->getOperand(1).isFI()) {
2884 FI = Def->getOperand(1).getIndex();
2885 Bytes = Flags.getByValSize();
2886 } else
2887 return false;
2888 }
2889 } else if (LoadSDNode *Ld = dyn_cast<LoadSDNode>(Arg)) {
2890 if (Flags.isByVal())
2891 // ByVal argument is passed in as a pointer but it's now being
2892 // dereferenced. e.g.
2893 // define @foo(%struct.X* %A) {
2894 // tail call @bar(%struct.X* byval %A)
2895 // }
2896 return false;
2897 SDValue Ptr = Ld->getBasePtr();
2899 if (!FINode)
2900 return false;
2901 FI = FINode->getIndex();
2902 } else if (Arg.getOpcode() == ISD::FrameIndex && Flags.isByVal()) {
2904 FI = FINode->getIndex();
2905 Bytes = Flags.getByValSize();
2906 } else
2907 return false;
2908
2909 assert(FI != INT_MAX);
2910 if (!MFI.isFixedObjectIndex(FI))
2911 return false;
2912
2913 if (Offset != MFI.getObjectOffset(FI))
2914 return false;
2915
2916 // If this is not byval, check that the argument stack object is immutable.
2917 // inalloca and argument copy elision can create mutable argument stack
2918 // objects. Byval objects can be mutated, but a byval call intends to pass the
2919 // mutated memory.
2920 if (!Flags.isByVal() && !MFI.isImmutableObjectIndex(FI))
2921 return false;
2922
2923 if (VA.getLocVT().getFixedSizeInBits() >
2925 // If the argument location is wider than the argument type, check that any
2926 // extension flags match.
2927 if (Flags.isZExt() != MFI.isObjectZExt(FI) ||
2928 Flags.isSExt() != MFI.isObjectSExt(FI)) {
2929 return false;
2930 }
2931 }
2932
2933 return Bytes == MFI.getObjectSize(FI);
2934}
2935
2936static bool
2938 Register CallerSRetReg) {
2939 const auto &Outs = CLI.Outs;
2940 const auto &OutVals = CLI.OutVals;
2941
2942 // We know the caller has a sret pointer argument (CallerSRetReg). Locate the
2943 // operand index within the callee that may have a sret pointer too.
2944 unsigned Pos = 0;
2945 for (unsigned E = Outs.size(); Pos != E; ++Pos)
2946 if (Outs[Pos].Flags.isSRet())
2947 break;
2948 // Bail out if the callee has not any sret argument.
2949 if (Pos == Outs.size())
2950 return false;
2951
2952 // At this point, either the caller is forwarding its sret argument to the
2953 // callee, or the callee is being passed a different sret pointer. We now look
2954 // for a CopyToReg, where the callee sret argument is written into a new vreg
2955 // (which should later be %rax/%eax, if this is returned).
2956 SDValue SRetArgVal = OutVals[Pos];
2957 for (SDNode *User : SRetArgVal->users()) {
2958 if (User->getOpcode() != ISD::CopyToReg)
2959 continue;
2961 if (Reg == CallerSRetReg && User->getOperand(2) == SRetArgVal)
2962 return true;
2963 }
2964
2965 return false;
2966}
2967
2968/// Check whether the call is eligible for sibling call optimization. Sibling
2969/// calls are loosely defined to be simple, profitable tail calls that only
2970/// require adjusting register parameters. We do not speculatively to optimize
2971/// complex calls that require lots of argument memory operations that may
2972/// alias.
2973///
2974/// Note that LLVM supports multiple ways, such as musttail, to force tail call
2975/// emission. Returning false from this function will not prevent tail call
2976/// emission in all cases.
2977bool X86TargetLowering::isEligibleForSiblingCallOpt(
2979 SmallVectorImpl<CCValAssign> &ArgLocs) const {
2980 SelectionDAG &DAG = CLI.DAG;
2981 const SmallVectorImpl<ISD::OutputArg> &Outs = CLI.Outs;
2982 const SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2983 const SmallVectorImpl<ISD::InputArg> &Ins = CLI.Ins;
2984 SDValue Callee = CLI.Callee;
2985 CallingConv::ID CalleeCC = CLI.CallConv;
2986 bool isVarArg = CLI.IsVarArg;
2987
2988 if (!mayTailCallThisCC(CalleeCC))
2989 return false;
2990
2991 // If -tailcallopt is specified, make fastcc functions tail-callable.
2992 MachineFunction &MF = DAG.getMachineFunction();
2993 X86MachineFunctionInfo *FuncInfo = MF.getInfo<X86MachineFunctionInfo>();
2994 const Function &CallerF = MF.getFunction();
2995
2996 // If the function return type is x86_fp80 and the callee return type is not,
2997 // then the FP_EXTEND of the call result is not a nop. It's not safe to
2998 // perform a tailcall optimization here.
2999 if (CallerF.getReturnType()->isX86_FP80Ty() && !CLI.RetTy->isX86_FP80Ty())
3000 return false;
3001
3002 // Win64 functions have extra shadow space for argument homing. Don't do the
3003 // sibcall if the caller and callee have mismatched expectations for this
3004 // space.
3005 CallingConv::ID CallerCC = CallerF.getCallingConv();
3006 bool IsCalleeWin64 = Subtarget.isCallingConvWin64(CalleeCC);
3007 bool IsCallerWin64 = Subtarget.isCallingConvWin64(CallerCC);
3008 if (IsCalleeWin64 != IsCallerWin64)
3009 return false;
3010
3011 // Do not optimize vararg calls with 6 arguments for LFI since LFI reserves
3012 // %r11, meaning there will not be enough registers available.
3013 if (Subtarget.isLFI() && ArgLocs.size() > 5)
3014 return false;
3015
3016 // If we are using a GOT, don't generate sibling calls to non-local,
3017 // default-visibility symbols. Tail calling such a symbol requires using a GOT
3018 // relocation, which forces early binding of the symbol. This breaks code that
3019 // require lazy function symbol resolution. Using musttail or
3020 // GuaranteedTailCallOpt will override this.
3021 if (Subtarget.isPICStyleGOT()) {
3022 if (isa<ExternalSymbolSDNode>(Callee))
3023 return false;
3024 if (GlobalAddressSDNode *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
3025 if (!G->getGlobal()->hasLocalLinkage() &&
3026 G->getGlobal()->hasDefaultVisibility())
3027 return false;
3028 }
3029 }
3030
3031 // Look for obvious safe cases to perform tail call optimization that do not
3032 // require ABI changes. This is what gcc calls sibcall.
3033
3034 // Can't do sibcall if stack needs to be dynamically re-aligned. PEI needs to
3035 // emit a special epilogue.
3036 const X86RegisterInfo *RegInfo = Subtarget.getRegisterInfo();
3037 if (RegInfo->hasStackRealignment(MF))
3038 return false;
3039
3040 // Avoid sibcall optimization if we are an sret return function and the callee
3041 // is incompatible, unless such premises are proven wrong. See comment in
3042 // LowerReturn about why hasStructRetAttr is insufficient.
3043 if (Register SRetReg = FuncInfo->getSRetReturnReg()) {
3044 // For a compatible tail call the callee must return our sret pointer. So it
3045 // needs to be (a) an sret function itself and (b) we pass our sret as its
3046 // sret. Condition #b is harder to determine.
3047 if (!mayBeSRetTailCallCompatible(CLI, SRetReg))
3048 return false;
3049 } else if (hasCalleePopSRet(Outs, ArgLocs, Subtarget))
3050 // The callee pops an sret, so we cannot tail-call, as our caller doesn't
3051 // expect that.
3052 return false;
3053
3054 // Do not sibcall optimize vararg calls unless all arguments are passed via
3055 // registers.
3056 LLVMContext &C = *DAG.getContext();
3057 if (isVarArg && !Outs.empty()) {
3058 // Optimizing for varargs on Win64 is unlikely to be safe without
3059 // additional testing.
3060 if (IsCalleeWin64 || IsCallerWin64)
3061 return false;
3062
3063 for (const auto &VA : ArgLocs)
3064 if (!VA.isRegLoc())
3065 return false;
3066 }
3067
3068 // If the call result is in ST0 / ST1, it needs to be popped off the x87
3069 // stack. Therefore, if it's not used by the call it is not safe to optimize
3070 // this into a sibcall.
3071 bool Unused = false;
3072 for (const auto &In : Ins) {
3073 if (!In.Used) {
3074 Unused = true;
3075 break;
3076 }
3077 }
3078 if (Unused) {
3080 CCState RVCCInfo(CalleeCC, false, MF, RVLocs, C);
3081 RVCCInfo.AnalyzeCallResult(Ins, RetCC_X86);
3082 for (const auto &VA : RVLocs) {
3083 if (VA.getLocReg() == X86::FP0 || VA.getLocReg() == X86::FP1)
3084 return false;
3085 }
3086 }
3087
3088 // Check that the call results are passed in the same way.
3089 if (!CCState::resultsCompatible(CalleeCC, CallerCC, MF, C, Ins,
3091 return false;
3092 // The callee has to preserve all registers the caller needs to preserve.
3093 const X86RegisterInfo *TRI = Subtarget.getRegisterInfo();
3094 const uint32_t *CallerPreserved = TRI->getCallPreservedMask(MF, CallerCC);
3095 if (CallerCC != CalleeCC) {
3096 const uint32_t *CalleePreserved = TRI->getCallPreservedMask(MF, CalleeCC);
3097 if (!TRI->regmaskSubsetEqual(CallerPreserved, CalleePreserved))
3098 return false;
3099 }
3100
3101 // The stack frame of the caller cannot be replaced by the tail-callee one's
3102 // if the function is required to preserve all the registers. Conservatively
3103 // prevent tail optimization even if hypothetically all the registers are used
3104 // for passing formal parameters or returning values.
3105 if (CallerF.hasFnAttribute("no_caller_saved_registers"))
3106 return false;
3107
3108 unsigned StackArgsSize = CCInfo.getStackSize();
3109
3110 // If the callee takes no arguments then go on to check the results of the
3111 // call.
3112 if (!Outs.empty()) {
3113 if (StackArgsSize > 0) {
3114 // Check if the arguments are already laid out in the right way as
3115 // the caller's fixed stack objects.
3116 MachineFrameInfo &MFI = MF.getFrameInfo();
3117 const MachineRegisterInfo *MRI = &MF.getRegInfo();
3118 const X86InstrInfo *TII = Subtarget.getInstrInfo();
3119 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
3120 const CCValAssign &VA = ArgLocs[I];
3121 SDValue Arg = OutVals[I];
3122 ISD::ArgFlagsTy Flags = Outs[I].Flags;
3124 return false;
3125 if (!VA.isRegLoc()) {
3126 if (!MatchingStackOffset(Arg, VA.getLocMemOffset(), Flags, MFI, MRI,
3127 TII, VA))
3128 return false;
3129 }
3130 }
3131 }
3132
3133 bool PositionIndependent = isPositionIndependent();
3134 // If the tailcall address may be in a register, then make sure it's
3135 // possible to register allocate for it. In 32-bit, the call address can
3136 // only target EAX, EDX, or ECX since the tail call must be scheduled after
3137 // callee-saved registers are restored. These happen to be the same
3138 // registers used to pass 'inreg' arguments so watch out for those.
3139 if (!Subtarget.is64Bit() && ((!isa<GlobalAddressSDNode>(Callee) &&
3140 !isa<ExternalSymbolSDNode>(Callee)) ||
3141 PositionIndependent)) {
3142 unsigned NumInRegs = 0;
3143 // In PIC we need an extra register to formulate the address computation
3144 // for the callee.
3145 unsigned MaxInRegs = PositionIndependent ? 2 : 3;
3146
3147 for (const auto &VA : ArgLocs) {
3148 if (!VA.isRegLoc())
3149 continue;
3150 Register Reg = VA.getLocReg();
3151 switch (Reg) {
3152 default: break;
3153 case X86::EAX: case X86::EDX: case X86::ECX:
3154 if (++NumInRegs == MaxInRegs)
3155 return false;
3156 break;
3157 }
3158 }
3159 }
3160
3161 const MachineRegisterInfo &MRI = MF.getRegInfo();
3162 if (!parametersInCSRMatch(MRI, CallerPreserved, ArgLocs, OutVals))
3163 return false;
3164 }
3165
3166 bool CalleeWillPop =
3167 X86::isCalleePop(CalleeCC, Subtarget.is64Bit(), isVarArg,
3169
3170 if (unsigned BytesToPop = FuncInfo->getBytesToPopOnReturn()) {
3171 // If we have bytes to pop, the callee must pop them.
3172 bool CalleePopMatches = CalleeWillPop && BytesToPop == StackArgsSize;
3173 if (!CalleePopMatches)
3174 return false;
3175 } else if (CalleeWillPop && StackArgsSize > 0) {
3176 // If we don't have bytes to pop, make sure the callee doesn't pop any.
3177 return false;
3178 }
3179
3180 return true;
3181}
3182
3183/// Determines whether the callee is required to pop its own arguments.
3184/// Callee pop is necessary to support tail calls.
3186 bool is64Bit, bool IsVarArg, bool GuaranteeTCO) {
3187 // If GuaranteeTCO is true, we force some calls to be callee pop so that we
3188 // can guarantee TCO.
3189 if (!IsVarArg && shouldGuaranteeTCO(CallingConv, GuaranteeTCO))
3190 return true;
3191
3192 switch (CallingConv) {
3193 default:
3194 return false;
3199 return !is64Bit;
3200 }
3201}
return SDValue()
static bool canGuaranteeTCO(CallingConv::ID CC, bool GuaranteeTailCalls)
Return true if the calling convention is one that we can guarantee TCO for.
static bool mayTailCallThisCC(CallingConv::ID CC)
Return true if we might ever do TCO for calls with this calling convention.
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
const HexagonInstrInfo * TII
static bool IsIndirectCall(const MachineInstr *MI)
static SDValue CreateCopyOfByValArgument(SDValue Src, SDValue Dst, SDValue Chain, ISD::ArgFlagsTy Flags, SelectionDAG &DAG, const SDLoc &dl)
CreateCopyOfByValArgument - Make a copy of an aggregate at address specified by "Src" to address "Dst...
Module.h This file contains the declarations for the Module class.
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
static LVOptions Options
Definition LVOptions.cpp:25
const MCPhysReg ArgGPRs[]
static bool shouldGuaranteeTCO(CallingConv::ID CC, bool GuaranteedTailCallOpt)
Return true if the function is being made into a tailcall target by changing its ABI.
static bool MatchingStackOffset(SDValue Arg, unsigned Offset, ISD::ArgFlagsTy Flags, MachineFrameInfo &MFI, const MachineRegisterInfo *MRI, const M68kInstrInfo *TII, const CCValAssign &VA)
Return true if the given stack call argument is already available in the same position (relatively) o...
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
Machine Check Debug Module
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
#define T
This file defines ARC utility functions which are used by various parts of the compiler.
static CodeModel::Model getCodeModel(const PPCSubtarget &S, const TargetMachine &TM, const MachineOperand &MO)
static void getMaxByValAlign(Type *Ty, Align &MaxAlign, Align MaxMaxAlign)
getMaxByValAlign - Helper for getByValTypeAlignment to determine the desired ByVal argument alignment...
const char * Msg
static bool contains(SmallPtrSetImpl< ConstantExpr * > &Cache, ConstantExpr *Expr, Constant *C)
Definition Value.cpp:484
This file defines the 'Statistic' class, which is designed to be an easy way to expose various metric...
#define STATISTIC(VARNAME, DESC)
Definition Statistic.h:171
static Function * getFunction(FunctionType *Ty, const Twine &Name, Module *M)
static bool is64Bit(const char *name)
static SDValue lowerMasksToReg(const SDValue &ValArg, const EVT &ValLoc, const SDLoc &DL, SelectionDAG &DAG)
Lowers masks values (v*i1) to the local register values.
static void Passv64i1ArgInRegs(const SDLoc &DL, SelectionDAG &DAG, SDValue &Arg, SmallVectorImpl< std::pair< Register, SDValue > > &RegsToPass, CCValAssign &VA, CCValAssign &NextVA, const X86Subtarget &Subtarget)
Breaks v64i1 value into two registers and adds the new node to the DAG.
static SDValue getv64i1Argument(CCValAssign &VA, CCValAssign &NextVA, SDValue &Root, SelectionDAG &DAG, const SDLoc &DL, const X86Subtarget &Subtarget, SDValue *InGlue=nullptr)
Reads two 32 bit registers and creates a 64 bit mask value.
static ArrayRef< MCPhysReg > get64BitArgumentXMMs(MachineFunction &MF, CallingConv::ID CallConv, const X86Subtarget &Subtarget)
static bool isSortedByValueNo(ArrayRef< CCValAssign > ArgLocs)
static ArrayRef< MCPhysReg > get64BitArgumentGPRs(CallingConv::ID CallConv, const X86Subtarget &Subtarget)
static SDValue getPopFromX87Reg(SelectionDAG &DAG, SDValue Chain, const SDLoc &dl, Register Reg, EVT VT, SDValue Glue)
static bool mayBeSRetTailCallCompatible(const TargetLowering::CallLoweringInfo &CLI, Register CallerSRetReg)
static std::pair< MVT, unsigned > handleMaskRegisterForCallingConv(unsigned NumElts, CallingConv::ID CC, const X86Subtarget &Subtarget)
static bool shouldDisableRetRegFromCSR(CallingConv::ID CC)
Returns true if a CC can dynamically exclude a register from the list of callee-saved-registers (Targ...
static void errorUnsupported(SelectionDAG &DAG, const SDLoc &dl, const char *Msg)
Call this when the user attempts to do something unsupported, like returning a double without SSE2 en...
static SDValue EmitTailCallStoreRetAddr(SelectionDAG &DAG, MachineFunction &MF, SDValue Chain, SDValue RetAddrFrIdx, EVT PtrVT, unsigned SlotSize, int FPDiff, const SDLoc &dl)
Emit a store of the return address if tail call optimization is performed and it is required (FPDiff!...
static bool shouldDisableArgRegFromCSR(CallingConv::ID CC)
Returns true if a CC can dynamically exclude a register from the list of callee-saved-registers (Targ...
static bool hasStackGuardSlotTLS(const Triple &TargetTriple)
static SDValue lowerRegToMasks(const SDValue &ValArg, const EVT &ValVT, const EVT &ValLoc, const SDLoc &DL, SelectionDAG &DAG)
The function will lower a register of various sizes (8/16/32/64) to a mask value of the expected size...
static Constant * SegmentOffset(IRBuilderBase &IRB, int Offset, unsigned AddressSpace)
static bool hasCalleePopSRet(const SmallVectorImpl< T > &Args, const SmallVectorImpl< CCValAssign > &ArgLocs, const X86Subtarget &Subtarget)
Determines whether Args, either a set of outgoing arguments to a call, or a set of incoming args of a...
static bool isBitAligned(Align Alignment, uint64_t SizeInBits)
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
size_t size() const
Get the array size.
Definition ArrayRef.h:141
ArrayRef< T > slice(size_t N, size_t M) const
slice(n, m) - Chop off the first N elements of the array, and keep M elements in the array.
Definition ArrayRef.h:185
const Function * getParent() const
Return the enclosing method, or null if none.
Definition BasicBlock.h:213
CCState - This class holds information needed while lowering arguments and return values.
static LLVM_ABI bool resultsCompatible(CallingConv::ID CalleeCC, CallingConv::ID CallerCC, MachineFunction &MF, LLVMContext &C, const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn CalleeFn, CCAssignFn CallerFn)
Returns true if the results of the two calling conventions are compatible.
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
CCValAssign - Represent assignment of one arg/retval to a location.
void convertToReg(MCRegister Reg)
Register getLocReg() const
LocInfo getLocInfo() const
bool needsCustom() const
bool isExtInLoc() const
int64_t getLocMemOffset() const
unsigned getValNo() const
CallingConv::ID getCallingConv() const
LLVM_ABI bool paramHasAttr(unsigned ArgNo, Attribute::AttrKind Kind) const
Determine whether the argument or parameter has the given attribute.
LLVM_ABI bool isMustTailCall() const
Tests if this call site must be tail call optimized.
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
static LLVM_ABI Constant * getIntToPtr(Constant *C, Type *Ty, bool OnlyIfReduced=false)
static ConstantInt * getSigned(IntegerType *Ty, int64_t V, bool ImplicitTrunc=false)
Return a ConstantInt with the specified value for the specified type.
Definition Constants.h:135
uint64_t getZExtValue() const
Return the constant as a 64-bit unsigned integer value after it has been zero extended as appropriate...
Definition Constants.h:168
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
LLVM_ABI TypeSize getTypeAllocSize(Type *Ty) const
Returns the offset in bytes between successive objects of the specified type, including alignment pad...
Diagnostic information for unsupported feature in backend.
A handy container for a FunctionType+Callee-pointer pair, which can be passed around as a single enti...
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:272
bool hasPersonalityFn() const
Check whether this function has a personality function.
Definition Function.h:882
Constant * getPersonalityFn() const
Get the personality function associated with this function.
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:723
const GlobalValue * getGlobal() const
Module * getParent()
Get the module that this global value is contained inside of...
void setDSOLocal(bool Local)
@ ExternalLinkage
Externally visible function.
Definition GlobalValue.h:53
Common base class shared among various IRBuilders.
Definition IRBuilder.h:114
BasicBlock * GetInsertBlock() const
Definition IRBuilder.h:175
LLVMContext & getContext() const
Definition IRBuilder.h:177
PointerType * getPtrTy(unsigned AddrSpace=0)
Fetch the type representing a pointer.
Definition IRBuilder.h:577
MDNode * getMetadata(unsigned KindID) const
Get the metadata of given kind attached to this Instruction.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
LLVM_ABI void diagnose(const DiagnosticInfo &DI)
Report a message to the currently installed diagnostic handler.
Tracks which library functions to use for a particular subtarget.
This class is used to represent ISD::LOAD nodes.
Context object for machine code objects.
Definition MCContext.h:83
Base class for the full range of assembler expressions which are needed for parsing.
Definition MCExpr.h:34
static const MCSymbolRefExpr * create(const MCSymbol *Symbol, MCContext &Ctx, SMLoc Loc=SMLoc())
Definition MCExpr.h:213
Machine Value Type.
@ INVALID_SIMPLE_VALUE_TYPE
SimpleValueType SimpleTy
unsigned getVectorNumElements() const
bool isVector() const
Return true if this is a vector value type.
bool is512BitVector() const
Return true if this is a 512-bit vector type.
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
MVT getVectorElementType() const
MVT getScalarType() const
If this is a vector, return the element type, otherwise return this.
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
void setObjectZExt(int ObjectIdx, bool IsZExt)
LLVM_ABI int CreateStackObject(uint64_t Size, Align Alignment, bool isSpillSlot, const AllocaInst *Alloca=nullptr, uint8_t ID=0)
Create a new statically sized stack object, returning a nonnegative identifier to represent it.
void setObjectSExt(int ObjectIdx, bool IsSExt)
bool isImmutableObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to an immutable object.
void setHasTailCall(bool V=true)
bool isObjectZExt(int ObjectIdx) const
int64_t getObjectSize(int ObjectIdx) const
Return the size of the specified object.
bool isObjectSExt(int ObjectIdx) const
int64_t getObjectOffset(int ObjectIdx) const
Return the assigned stack offset of the specified object from the incoming stack pointer.
bool isFixedObjectIndex(int ObjectIdx) const
Returns true if the specified index corresponds to a fixed stack object.
int getObjectIndexBegin() const
Return the minimum frame object index.
const WinEHFuncInfo * getWinEHFuncInfo() const
getWinEHFuncInfo - Return information about how the current function uses Windows exception handling.
MCSymbol * getPICBaseSymbol() const
getPICBaseSymbol - Return a function-local symbol to represent the PIC base.
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags f, LLT MemTy, Align base_alignment, const AAMDNodes &AAInfo=AAMDNodes(), const MDNode *Ranges=nullptr, SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
uint32_t * allocateRegMask()
Allocate and initialize a register mask with NumRegister bits.
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
const TargetMachine & getTarget() const
getTarget - Return the target machine this machine code is compiled with
Representation of each machine instruction.
@ EK_Custom32
EK_Custom32 - Each entry is a 32-bit value that is custom lowered by the TargetLowering::LowerCustomJ...
@ EK_LabelDifference64
EK_LabelDifference64 - Each entry is the address of the block minus the address of the jump table.
A description of a memory reference used in the backend.
Flags
Flags values. These may be or'd together.
@ MOLoad
The memory access reads data.
@ MONonTemporal
The memory access is non-temporal.
@ MOStore
The memory access writes data.
static unsigned getRegMaskSize(unsigned NumRegs)
Returns number of elements needed for a regmask array.
static bool clobbersPhysReg(const uint32_t *RegMask, MCRegister PhysReg)
clobbersPhysReg - Returns true if this RegMask clobbers PhysReg.
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
LLVM_ABI MachineInstr * getVRegDef(Register Reg) const
getVRegDef - Return the machine instr that defines the specified virtual register or null if none is ...
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
ArrayRef< std::pair< MCRegister, Register > > liveins() const
LLVM_ABI void disableCalleeSavedRegister(MCRegister Reg)
Disables the register from the list of CSRs.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:67
static PointerType * getUnqual(Type *ElementType)
This constructs a pointer to an object of the specified type in the default address space (address sp...
Wrapper class representing virtual and physical registers.
Definition Register.h:20
constexpr bool isVirtual() const
Return true if the specified register number is in the virtual register namespace.
Definition Register.h:79
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
const DebugLoc & getDebugLoc() const
Represents one node in the SelectionDAG.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
void setCFIType(uint32_t Type)
iterator_range< user_iterator > users()
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
SDNode * getNode() const
get the SDNode which holds the desired result
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getStackArgumentTokenFactor(SDValue Chain)
Compute a TokenFactor to force all the incoming stack arguments to be loaded from the stack.
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const AAMDNodes &AAInfo=AAMDNodes(), const MDNode *Ranges=nullptr)
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
void addHeapAllocSite(const SDNode *Node, MDNode *MD)
Set HeapAllocSite to be associated with Node.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const AAMDNodes &AAInfo=AAMDNodes())
Helper function to build ISD::STORE nodes.
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
const TargetMachine & getTarget() const
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
MachineFunction & getMachineFunction() const
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
void addCallSiteInfo(const SDNode *Node, CallSiteInfo &&CallInfo)
Set CallSiteInfo to be associated with Node.
LLVMContext * getContext() const
LLVM_ABI SDValue CreateStackTemporary(TypeSize Bytes, Align Alignment)
Create a stack temporary based on the size in bytes and the alignment.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
void assign(size_type NumElts, ValueParamT Elt)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
Class to represent struct types.
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
virtual Value * getIRStackGuard(IRBuilderBase &IRB, const LibcallLoweringInfo &Libcalls) const
If the target has a standard location for the stack protector guard, returns the address of that loca...
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
virtual void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const
Inserts necessary declarations for SSP (stack protection) purpose.
virtual unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
virtual std::pair< const TargetRegisterClass *, uint8_t > findRepresentativeClass(const TargetRegisterInfo *TRI, MVT VT) const
Return the largest legal super-reg register class of the register class for the specified type and it...
static StringRef getLibcallImplName(RTLIB::LibcallImpl Call)
Get the libcall routine name for the specified libcall implementation.
virtual Value * getSafeStackPointerLocation(IRBuilderBase &IRB, const LibcallLoweringInfo &Libcalls) const
Returns the target-specific address of the unsafe stack pointer.
LegalizeTypeAction getTypeAction(LLVMContext &Context, EVT VT) const
Return how we should legalize values of this type, either it is already legal (return 'Legal') or we ...
std::vector< ArgListEntry > ArgListTy
MVT getRegisterType(MVT VT) const
Return the type of registers that this ValueType will eventually require.
virtual const MCExpr * getPICJumpTableRelocBaseExpr(const MachineFunction *MF, unsigned JTI, MCContext &Ctx) const
This returns the relocation base for the given PIC jumptable, the same as getPICJumpTableRelocBase,...
bool parametersInCSRMatch(const MachineRegisterInfo &MRI, const uint32_t *CallerPreservedMask, const SmallVectorImpl< CCValAssign > &ArgLocs, const SmallVectorImpl< SDValue > &OutVals) const
Check whether parameters to a call that are passed in callee saved registers are the same as from the...
bool isPositionIndependent() const
virtual ArrayRef< MCPhysReg > getRoundingControlRegisters() const
Returns a 0 terminated array of rounding control registers that can be attached into strict FP call.
virtual unsigned getJumpTableEncoding() const
Return the entry encoding for a jump table in the current function.
void setTypeIdForCallsiteInfo(const CallBase *CB, MachineFunction &MF, MachineFunction::CallSiteInfo &CSInfo) const
TargetOptions Options
CodeModel::Model getCodeModel() const
Returns the code model.
unsigned GuaranteedTailCallOpt
GuaranteedTailCallOpt - This flag is enabled when -tailcallopt is specified on the commandline.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
Triple - Helper class for working with autoconf configuration names.
Definition Triple.h:47
bool isAndroid() const
Tests whether the target is Android.
Definition Triple.h:907
bool isMusl() const
Tests whether the environment is musl-libc.
Definition Triple.h:922
bool isOSGlibc() const
Tests whether the OS uses glibc.
Definition Triple.h:845
bool isOSFuchsia() const
Definition Triple.h:747
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
static LLVM_ABI IntegerType * getInt64Ty(LLVMContext &C)
Definition Type.cpp:310
bool isX86_FP80Ty() const
Return true if this is x86 long double.
Definition Type.h:161
static LLVM_ABI IntegerType * getInt32Ty(LLVMContext &C)
Definition Type.cpp:309
static LLVM_ABI Type * getVoidTy(LLVMContext &C)
Definition Type.cpp:282
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
void setBytesToPopOnReturn(unsigned bytes)
void setVarArgsGPOffset(unsigned Offset)
SmallVectorImpl< ForwardedRegister > & getForwardedMustTailRegParms()
void setVarArgsFPOffset(unsigned Offset)
const uint32_t * getCallPreservedMask(const MachineFunction &MF, CallingConv::ID) const override
Register getStackRegister() const
unsigned getSlotSize() const
Register getFramePtr() const
Returns physical register used as frame pointer.
Register getBaseRegister() const
const uint32_t * getNoPreservedMask() const override
bool hasSSE1() const
const Triple & getTargetTriple() const
bool useAVX512Regs() const
bool isCallingConvWin64(CallingConv::ID CC) const
bool isOSWindowsOrUEFI() const
std::pair< const TargetRegisterClass *, uint8_t > findRepresentativeClass(const TargetRegisterInfo *TRI, MVT VT) const override
Return the largest legal super-reg register class of the register class for the specified type and it...
SDValue getPICJumpTableRelocBase(SDValue Table, SelectionDAG &DAG) const override
Returns relocation base for the given PIC jumptable.
unsigned getJumpTableEncoding() const override
Return the entry encoding for a jump table in the current function.
bool isMemoryAccessFast(EVT VT, Align Alignment) const
Value * getIRStackGuard(IRBuilderBase &IRB, const LibcallLoweringInfo &Libcalls) const override
If the target has a standard location for the stack protector cookie, returns the address of that loc...
bool useSoftFloat() const override
Value * getSafeStackPointerLocation(IRBuilderBase &IRB, const LibcallLoweringInfo &Libcalls) const override
Return true if the target stores SafeStack pointer at a fixed offset in some non-standard address spa...
const MCExpr * getPICJumpTableRelocBaseExpr(const MachineFunction *MF, unsigned JTI, MCContext &Ctx) const override
This returns the relocation base for the given PIC jumptable, the same as getPICJumpTableRelocBase,...
bool isSafeMemOpType(MVT VT) const override
Returns true if it's safe to use load / store of the specified type to expand memcpy / memset inline.
bool functionArgumentNeedsConsecutiveRegisters(Type *Ty, CallingConv::ID CallConv, bool isVarArg, const DataLayout &DL) const override
For some targets, an LLVM struct type must be broken down into multiple simple types,...
Align getByValTypeAlignment(Type *Ty, const DataLayout &DL) const override
Return the desired alignment for ByVal aggregate function arguments in the caller parameter area.
MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AS, Align Alignment, MachineMemOperand::Flags Flags, unsigned *Fast) const override
Returns true if the target allows unaligned memory accesses of the specified type.
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
It returns EVT::Other if the type should be determined using generic target-independent logic.
unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
void markLibCallAttributes(MachineFunction *MF, unsigned CC, ArgListTy &Args) const override
void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const override
Inserts necessary declarations for SSP (stack protection) purpose.
bool isScalarFPTypeInSSEReg(EVT VT) const
Return true if the specified scalar FP type is computed in an SSE register, not on the X87 floating p...
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
bool allowsMemoryAccess(LLVMContext &Context, const DataLayout &DL, EVT VT, unsigned AddrSpace, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MONone, unsigned *Fast=nullptr) const override
This function returns true if the memory access is aligned or if the target allows this specific unal...
SDValue getReturnAddressFrameIndex(SelectionDAG &DAG) const
SDValue unwrapAddress(SDValue N) const override
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &Context, EVT VT) const override
Return the value type to use for ISD::SETCC.
EVT getTypeToTransformTo(LLVMContext &Context, EVT VT) const override
For types supported by the target, this is an identity function.
const MCExpr * LowerCustomJumpTableEntry(const MachineJumpTableInfo *MJTI, const MachineBasicBlock *MBB, unsigned uid, MCContext &Ctx) const override
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
CallingConv Namespace - This namespace contains an enum with a value for the well-known calling conve...
Definition CallingConv.h:21
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ X86_64_SysV
The C convention as specified in the x86-64 supplement to the System V ABI, used on most non-Windows ...
@ HiPE
Used by the High-Performance Erlang Compiler (HiPE).
Definition CallingConv.h:53
@ Swift
Calling convention for Swift.
Definition CallingConv.h:69
@ PreserveMost
Used for runtime calls that preserves most registers.
Definition CallingConv.h:63
@ X86_INTR
x86 hardware interrupt context.
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ X86_ThisCall
Similar to X86_StdCall.
@ PreserveAll
Used for runtime calls that preserves (almost) all registers.
Definition CallingConv.h:66
@ X86_StdCall
stdcall is mostly used by the Win32 API.
Definition CallingConv.h:99
@ Fast
Attempts to make calls as fast as possible (e.g.
Definition CallingConv.h:41
@ X86_VectorCall
MSVC calling convention that passes vectors and vector aggregates in SSE registers.
@ Intel_OCL_BI
Used for Intel OpenCL built-ins.
@ PreserveNone
Used for runtime calls that preserves none general registers.
Definition CallingConv.h:90
@ Tail
Attemps to make calls as fast as possible while guaranteeing that tail call optimization can always b...
Definition CallingConv.h:76
@ Win64
The C convention as implemented on Windows/x86-64 and AArch64.
@ SwiftTail
This follows the Swift calling convention in how arguments are passed but guarantees tail calls will ...
Definition CallingConv.h:87
@ X86_RegCall
Register calling convention used for parameters transfer optimization.
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
@ X86_FastCall
'fast' analog of X86_StdCall.
NodeType
ISD::NodeType enum - This enum defines the target-independent operators for a SelectionDAG.
Definition ISDOpcodes.h:41
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:264
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:863
@ GlobalAddress
Definition ISDOpcodes.h:88
@ CONCAT_VECTORS
CONCAT_VECTORS(VECTOR0, VECTOR1, ...) - Given a number of values of vector type with the same length ...
Definition ISDOpcodes.h:586
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:854
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:667
@ CopyFromReg
CopyFromReg - This node indicates that the input value is a virtual or physical register that is defi...
Definition ISDOpcodes.h:230
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:578
@ CopyToReg
CopyToReg - This node has three operands: a chain, a register number to set to this value,...
Definition ISDOpcodes.h:224
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:860
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:988
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:53
@ ExternalSymbol
Definition ISDOpcodes.h:93
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:969
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:866
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:62
LLVM_ABI LegalityPredicate isVector(unsigned TypeIdx)
True iff the specified type index is a vector.
@ MO_NO_FLAG
MO_NO_FLAG - No flag for the operand.
@ GlobalBaseReg
On Darwin, this node represents the result of the popl at function entry, used for PIC code.
@ POP_FROM_X87_REG
The same as ISD::CopyFromReg except that this node makes it explicit that it may lower to an x87 FPU ...
bool isExtendedSwiftAsyncFrameSupported(const X86Subtarget &Subtarget, const MachineFunction &MF)
True if the target supports the extended frame for async Swift functions.
bool isCalleePop(CallingConv::ID CallingConv, bool is64Bit, bool IsVarArg, bool GuaranteeTCO)
Determines whether the callee is required to pop its own arguments.
std::optional< Function * > getAttachedARCFunction(const CallBase *CB)
This function returns operand bundle clang_arc_attachedcall's argument, which is the address of the A...
Definition ObjCARCUtil.h:43
bool hasAttachedCallOpBundle(const CallBase *CB)
Definition ObjCARCUtil.h:29
This is an optimization pass for GlobalISel generic memory operations.
@ Offset
Definition DWP.cpp:578
LLVM_ABI bool isCFGuardCall(const CallBase *CB)
Definition CFGuard.cpp:318
InstructionCost Cost
@ Unknown
Not known to have no common set bits.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
LLVM_ABI bool isCFGuardFunction(const GlobalValue *GV)
Definition CFGuard.cpp:323
@ Store
The extracted value is stored (ExtractElement only).
void append_range(Container &C, Range &&R)
Wrapper function to append range R to container C.
Definition STLExtras.h:2208
bool any_of(R &&range, UnaryPredicate P)
Provide wrappers to std::any_of which take ranges instead of having to pass begin/end explicitly.
Definition STLExtras.h:1746
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr uint64_t alignTo(uint64_t Size, Align A)
Returns a multiple of A needed to store Size bytes.
Definition Alignment.h:144
bool is_sorted(R &&Range, Compare C)
Wrapper function around std::is_sorted to check if elements in a range R are sorted with respect to a...
Definition STLExtras.h:1970
LLVM_ABI EHPersonality classifyEHPersonality(const Value *Pers)
See if the given exception handling personality function is one that we understand.
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
bool isFuncletEHPersonality(EHPersonality Pers)
Returns true if this is a personality function that invokes handler funclets (which must return to it...
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
bool CC_X86(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
bool RetCC_X86(unsigned ValNo, MVT ValVT, MVT LocVT, CCValAssign::LocInfo LocInfo, ISD::ArgFlagsTy ArgFlags, Type *OrigTy, CCState &State)
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
LLVM_ABI void reportFatalUsageError(Error Err)
Report a fatal error that does not indicate a bug in LLVM.
Definition Error.cpp:177
#define N
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
constexpr uint64_t value() const
This is a hole in the type system and should not be abused.
Definition Alignment.h:77
static constexpr Align Constant()
Allow constructions of constexpr Align.
Definition Alignment.h:88
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool bitsLT(EVT VT) const
Return true if this has less bits than VT.
Definition ValueTypes.h:323
ElementCount getVectorElementCount() const
Definition ValueTypes.h:373
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
EVT changeVectorElementType(LLVMContext &Context, EVT EltVT) const
Return a VT for a vector type whose attributes match ourselves with the exception of the element type...
Definition ValueTypes.h:98
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
bool is128BitVector() const
Return true if this is a 128-bit vector type.
Definition ValueTypes.h:230
bool is512BitVector() const
Return true if this is a 512-bit vector type.
Definition ValueTypes.h:240
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
bool is256BitVector() const
Return true if this is a 256-bit vector type.
Definition ValueTypes.h:235
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isVectorOf(EVT EltVT) const
Return true if this is a vector with matching element type.
Definition ValueTypes.h:181
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
Describes a register that needs to be forwarded from the prologue to a musttail call.
SmallVector< ArgRegPair, 1 > ArgRegPairs
Vector of call argument and its forwarding register.
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getStack(MachineFunction &MF, int64_t Offset, uint8_t ID=0)
Stack pointer relative access.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
SmallVector< ISD::OutputArg, 32 > Outs
Type * RetTy
Same as OrigRetTy, or partially legalized for soft float libcalls.