LLVM 24.0.0git
SystemZISelLowering.cpp
Go to the documentation of this file.
1//===-- SystemZISelLowering.cpp - SystemZ DAG lowering implementation -----===//
2//
3// Part of the LLVM Project, under the Apache License v2.0 with LLVM Exceptions.
4// See https://llvm.org/LICENSE.txt for license information.
5// SPDX-License-Identifier: Apache-2.0 WITH LLVM-exception
6//
7//===----------------------------------------------------------------------===//
8//
9// This file implements the SystemZTargetLowering class.
10//
11//===----------------------------------------------------------------------===//
12
13#include "SystemZISelLowering.h"
14#include "SystemZCallingConv.h"
17#include "llvm/ADT/SmallSet.h"
22#include "llvm/IR/GlobalAlias.h"
24#include "llvm/IR/Intrinsics.h"
25#include "llvm/IR/IntrinsicsS390.h"
26#include "llvm/IR/Module.h"
32#include <cctype>
33#include <optional>
34
35using namespace llvm;
36
37#define DEBUG_TYPE "systemz-lower"
38
39// Temporarily let this be disabled by default until all known problems
40// related to argument extensions are fixed.
42 "argext-abi-check", cl::init(false),
43 cl::desc("Verify that narrow int args are properly extended per the "
44 "SystemZ ABI."));
45
46namespace {
47// Represents information about a comparison.
48struct Comparison {
49 Comparison(SDValue Op0In, SDValue Op1In, SDValue ChainIn)
50 : Op0(Op0In), Op1(Op1In), Chain(ChainIn),
51 Opcode(0), ICmpType(0), CCValid(0), CCMask(0) {}
52
53 // The operands to the comparison.
54 SDValue Op0, Op1;
55
56 // Chain if this is a strict floating-point comparison.
57 SDValue Chain;
58
59 // The opcode that should be used to compare Op0 and Op1.
60 unsigned Opcode;
61
62 // A SystemZICMP value. Only used for integer comparisons.
63 unsigned ICmpType;
64
65 // The mask of CC values that Opcode can produce.
66 unsigned CCValid;
67
68 // The mask of CC values for which the original condition is true.
69 unsigned CCMask;
70};
71} // end anonymous namespace
72
73// Classify VT as either 32 or 64 bit.
74static bool is32Bit(EVT VT) {
75 switch (VT.getSimpleVT().SimpleTy) {
76 case MVT::i32:
77 return true;
78 case MVT::i64:
79 return false;
80 default:
81 llvm_unreachable("Unsupported type");
82 }
83}
84
85// Return a version of MachineOperand that can be safely used before the
86// final use.
88 if (Op.isReg())
89 Op.setIsKill(false);
90 return Op;
91}
92
94 const SystemZSubtarget &STI)
95 : TargetLowering(TM, STI), Subtarget(STI) {
96 MVT PtrVT = MVT::i64;
97
98 auto *Regs = STI.getSpecialRegisters();
99
100 // Set up the register classes.
101 if (Subtarget.hasHighWord())
102 addRegisterClass(MVT::i32, &SystemZ::GRX32BitRegClass);
103 else
104 addRegisterClass(MVT::i32, &SystemZ::GR32BitRegClass);
105 addRegisterClass(MVT::i64, &SystemZ::GR64BitRegClass);
106 if (!useSoftFloat()) {
107 if (Subtarget.hasVector()) {
108 addRegisterClass(MVT::f16, &SystemZ::VR16BitRegClass);
109 addRegisterClass(MVT::f32, &SystemZ::VR32BitRegClass);
110 addRegisterClass(MVT::f64, &SystemZ::VR64BitRegClass);
111 } else {
112 addRegisterClass(MVT::f16, &SystemZ::FP16BitRegClass);
113 addRegisterClass(MVT::f32, &SystemZ::FP32BitRegClass);
114 addRegisterClass(MVT::f64, &SystemZ::FP64BitRegClass);
115 }
116 if (Subtarget.hasVectorEnhancements1())
117 addRegisterClass(MVT::f128, &SystemZ::VR128BitRegClass);
118 else
119 addRegisterClass(MVT::f128, &SystemZ::FP128BitRegClass);
120
121 if (Subtarget.hasVector()) {
122 addRegisterClass(MVT::v16i8, &SystemZ::VR128BitRegClass);
123 addRegisterClass(MVT::v8i16, &SystemZ::VR128BitRegClass);
124 addRegisterClass(MVT::v4i32, &SystemZ::VR128BitRegClass);
125 addRegisterClass(MVT::v2i64, &SystemZ::VR128BitRegClass);
126 addRegisterClass(MVT::v8f16, &SystemZ::VR128BitRegClass);
127 addRegisterClass(MVT::v4f32, &SystemZ::VR128BitRegClass);
128 addRegisterClass(MVT::v2f64, &SystemZ::VR128BitRegClass);
129 }
130
131 if (Subtarget.hasVector())
132 addRegisterClass(MVT::i128, &SystemZ::VR128BitRegClass);
133 }
134
135 // Compute derived properties from the register classes
136 computeRegisterProperties(Subtarget.getRegisterInfo());
137
138 // Set up special registers.
139 setStackPointerRegisterToSaveRestore(Regs->getStackPointerRegister());
140
141 // TODO: It may be better to default to latency-oriented scheduling, however
142 // LLVM's current latency-oriented scheduler can't handle physreg definitions
143 // such as SystemZ has with CC, so set this to the register-pressure
144 // scheduler, because it can.
146
149
151
152 // Instructions are strings of 2-byte aligned 2-byte values.
154 // For performance reasons we prefer 16-byte alignment.
156
157 // Handle operations that are handled in a similar way for all types.
158 for (unsigned I = MVT::FIRST_INTEGER_VALUETYPE;
159 I <= MVT::LAST_FP_VALUETYPE;
160 ++I) {
162 if (isTypeLegal(VT)) {
163 // Lower SET_CC into an IPM-based sequence.
167
168 // Expand SELECT(C, A, B) into SELECT_CC(X, 0, A, B, NE).
170
171 // Lower SELECT_CC and BR_CC into separate comparisons and branches.
174 }
175 }
176
177 // Expand jump table branches as address arithmetic followed by an
178 // indirect jump.
180
181 // Expand BRCOND into a BR_CC (see above).
183
184 // Handle integer types except i128.
185 for (unsigned I = MVT::FIRST_INTEGER_VALUETYPE;
186 I <= MVT::LAST_INTEGER_VALUETYPE;
187 ++I) {
189 if (isTypeLegal(VT) && VT != MVT::i128) {
191
192 // Expand individual DIV and REMs into DIVREMs.
199
200 // Support addition/subtraction with overflow.
203
204 // Support addition/subtraction with carry.
207
208 // Support carry in as value rather than glue.
211
212 // Lower ATOMIC_LOAD_SUB into ATOMIC_LOAD_ADD if LAA and LAAG are
213 // available, or if the operand is constant.
215
216 // Use POPCNT on z196 and above.
217 if (Subtarget.hasPopulationCount())
219 else
221
222 // No special instructions for these.
225
226 // Use *MUL_LOHI where possible instead of MULH*.
231
232 // The fp<=>i32/i64 conversions are all Legal except for f16 and for
233 // unsigned on z10 (only z196 and above have native support for
234 // unsigned conversions).
241 // Handle unsigned 32-bit input types as signed 64-bit types on z10.
242 auto OpAction =
243 (!Subtarget.hasFPExtension() && VT == MVT::i32) ? Promote : Custom;
244 setOperationAction(Op, VT, OpAction);
245 }
246 }
247 }
248
249 // Handle i128 if legal.
250 if (isTypeLegal(MVT::i128)) {
251 // No special instructions for these.
258
259 // We may be able to use VSLDB/VSLD/VSRD for these.
262
263 // No special instructions for these before z17.
264 if (!Subtarget.hasVectorEnhancements3()) {
274 } else {
275 // Even if we do have a legal 128-bit multiply, we do not
276 // want 64-bit multiply-high operations to use it.
279 }
280
281 // Support addition/subtraction with carry.
286
287 // Use VPOPCT and add up partial results.
289
290 // Additional instructions available with z17.
291 if (Subtarget.hasVectorEnhancements3()) {
292 setOperationAction(ISD::ABS, MVT::i128, Legal);
293
295 MVT::i128, Legal);
296 }
297 }
298
299 // These need custom handling in order to handle the f16 conversions.
308
309 // Type legalization will convert 8- and 16-bit atomic operations into
310 // forms that operate on i32s (but still keeping the original memory VT).
311 // Lower them into full i32 operations.
323
324 // Whether or not i128 is not a legal type, we need to custom lower
325 // the atomic operations in order to exploit SystemZ instructions.
330
331 // Mark sign/zero extending atomic loads as legal, which will make
332 // DAGCombiner fold extensions into atomic loads if possible.
334 {MVT::i8, MVT::i16, MVT::i32}, Legal);
336 {MVT::i8, MVT::i16}, Legal);
338 MVT::i8, Legal);
339
340 // We can use the CC result of compare-and-swap to implement
341 // the "success" result of ATOMIC_CMP_SWAP_WITH_SUCCESS.
345
347
348 // Traps are legal, as we will convert them to "j .+2".
349 setOperationAction(ISD::TRAP, MVT::Other, Legal);
350
351 // We have native support for a 64-bit CTLZ, via FLOGR.
355
356 // On z17 we have native support for a 64-bit CTTZ.
357 if (Subtarget.hasMiscellaneousExtensions4()) {
361 }
362
363 // On z15 we have native support for a 64-bit CTPOP.
364 if (Subtarget.hasMiscellaneousExtensions3()) {
367 }
368
369 // Give LowerOperation the chance to replace 64-bit ORs with subregs.
371
372 // Expand 128 bit shifts without using a libcall.
376
377 // Also expand 256 bit shifts if i128 is a legal type.
378 if (isTypeLegal(MVT::i128)) {
382 }
383
384 // Handle bitcast from fp128 to i128.
385 if (!isTypeLegal(MVT::i128))
387
388 // We have native instructions for i8, i16 and i32 extensions, but not i1.
390 for (MVT VT : MVT::integer_valuetypes()) {
394 }
395
396 // Handle the various types of symbolic address.
402
403 // We need to handle dynamic allocations specially because of the
404 // 160-byte area at the bottom of the stack.
407
410
411 // Handle prefetches with PFD or PFDRL.
413
414 // Handle readcyclecounter with STCKF.
416
418 // Assume by default that all vector operations need to be expanded.
419 for (unsigned Opcode = 0; Opcode < ISD::BUILTIN_OP_END; ++Opcode)
420 if (getOperationAction(Opcode, VT) == Legal)
421 setOperationAction(Opcode, VT, Expand);
422
423 // Likewise all truncating stores and extending loads.
424 for (MVT InnerVT : MVT::fixedlen_vector_valuetypes()) {
425 setTruncStoreAction(VT, InnerVT, Expand);
428 setLoadExtAction(ISD::EXTLOAD, VT, InnerVT, Expand);
429 }
430
431 if (isTypeLegal(VT)) {
432 // These operations are legal for anything that can be stored in a
433 // vector register, even if there is no native support for the format
434 // as such. In particular, we can do these for v4f32 even though there
435 // are no specific instructions for that format.
441
442 // Likewise, except that we need to replace the nodes with something
443 // more specific.
446 }
447 }
448
449 // Handle integer vector types.
451 if (isTypeLegal(VT)) {
452 // These operations have direct equivalents.
457 if (VT != MVT::v2i64 || Subtarget.hasVectorEnhancements3()) {
461 }
462 if (Subtarget.hasVectorEnhancements3() &&
463 VT != MVT::v16i8 && VT != MVT::v8i16) {
468 }
473 if (Subtarget.hasVectorEnhancements1())
475 else
479
480 // Convert a GPR scalar to a vector by inserting it into element 0.
482
483 // Use a series of unpacks for extensions.
486
487 // Detect shifts/rotates by a scalar amount and convert them into
488 // V*_BY_SCALAR.
493
494 // Add ISD::VECREDUCE_ADD as custom in order to implement
495 // it with VZERO+VSUM
497
498 // Map SETCCs onto one of VCE, VCH or VCHL, swapping the operands
499 // and inverting the result as necessary.
501
503 Legal);
504 }
505 }
506
507 if (Subtarget.hasVector()) {
508 // There should be no need to check for float types other than v2f64
509 // since <2 x f32> isn't a legal type.
518
527 }
528
529 if (Subtarget.hasVectorEnhancements2()) {
538
547 }
548
549 // Handle floating-point types.
550 if (!useSoftFloat()) {
551 // Promote all f16 operations to float, with some exceptions below.
552 for (unsigned Opc = 0; Opc < ISD::BUILTIN_OP_END; ++Opc)
553 setOperationAction(Opc, MVT::f16, Promote);
555 for (MVT VT : {MVT::f32, MVT::f64, MVT::f128}) {
556 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f16, Expand);
557 setTruncStoreAction(VT, MVT::f16, Expand);
558 }
560 setOperationAction(Op, MVT::f16, Subtarget.hasVector() ? Legal : Custom);
564
565 for (auto Op : {ISD::FNEG, ISD::FABS, ISD::FCOPYSIGN})
566 setOperationAction(Op, MVT::f16, Legal);
567 }
568
569 for (unsigned I = MVT::FIRST_FP_VALUETYPE;
570 I <= MVT::LAST_FP_VALUETYPE;
571 ++I) {
573 if (isTypeLegal(VT) && VT != MVT::f16) {
574 // We can use FI for FRINT.
576
577 // We can use the extended form of FI for other rounding operations.
578 if (Subtarget.hasFPExtension()) {
585 }
586
587 // No special instructions for these.
593
594 // Special treatment.
596
597 // Handle constrained floating-point operations.
606 if (Subtarget.hasFPExtension()) {
613 }
614
615 // Extension from f16 needs libcall.
618 }
619 }
620
621 // Handle floating-point vector types.
622 if (Subtarget.hasVector()) {
623 // Scalar-to-vector conversion is just a subreg.
627
628 // Some insertions and extractions can be done directly but others
629 // need to go via integers.
636
637 // These operations have direct equivalents.
638 setOperationAction(ISD::FADD, MVT::v2f64, Legal);
639 setOperationAction(ISD::FNEG, MVT::v2f64, Legal);
640 setOperationAction(ISD::FSUB, MVT::v2f64, Legal);
641 setOperationAction(ISD::FMUL, MVT::v2f64, Legal);
642 setOperationAction(ISD::FMA, MVT::v2f64, Legal);
643 setOperationAction(ISD::FDIV, MVT::v2f64, Legal);
644 setOperationAction(ISD::FABS, MVT::v2f64, Legal);
645 setOperationAction(ISD::FSQRT, MVT::v2f64, Legal);
646 setOperationAction(ISD::FRINT, MVT::v2f64, Legal);
649 setOperationAction(ISD::FCEIL, MVT::v2f64, Legal);
653
654 // Handle constrained floating-point operations.
668
673 if (Subtarget.hasVectorEnhancements1()) {
676 }
677 }
678
679 // The vector enhancements facility 1 has instructions for these.
680 if (Subtarget.hasVectorEnhancements1()) {
681 setOperationAction(ISD::FADD, MVT::v4f32, Legal);
682 setOperationAction(ISD::FNEG, MVT::v4f32, Legal);
683 setOperationAction(ISD::FSUB, MVT::v4f32, Legal);
684 setOperationAction(ISD::FMUL, MVT::v4f32, Legal);
685 setOperationAction(ISD::FMA, MVT::v4f32, Legal);
686 setOperationAction(ISD::FDIV, MVT::v4f32, Legal);
687 setOperationAction(ISD::FABS, MVT::v4f32, Legal);
688 setOperationAction(ISD::FSQRT, MVT::v4f32, Legal);
689 setOperationAction(ISD::FRINT, MVT::v4f32, Legal);
692 setOperationAction(ISD::FCEIL, MVT::v4f32, Legal);
696
697 for (MVT Type : {MVT::f64, MVT::v2f64, MVT::f32, MVT::v4f32, MVT::f128}) {
706 }
707
708 // Handle constrained floating-point operations.
722 for (auto VT : { MVT::f32, MVT::f64, MVT::f128,
723 MVT::v4f32, MVT::v2f64 }) {
730 }
731 }
732
733 // We only have fused f128 multiply-addition on vector registers.
734 if (!Subtarget.hasVectorEnhancements1()) {
737 }
738
739 // We don't have a copysign instruction on vector registers.
740 if (Subtarget.hasVectorEnhancements1())
742
743 // Needed so that we don't try to implement f128 constant loads using
744 // a load-and-extend of a f80 constant (in cases where the constant
745 // would fit in an f80).
746 for (MVT VT : MVT::fp_valuetypes())
747 setLoadExtAction(ISD::EXTLOAD, VT, MVT::f80, Expand);
748
749 // We don't have extending load instruction on vector registers.
750 if (Subtarget.hasVectorEnhancements1()) {
751 setLoadExtAction(ISD::EXTLOAD, MVT::f128, MVT::f32, Expand);
752 setLoadExtAction(ISD::EXTLOAD, MVT::f128, MVT::f64, Expand);
753 }
754
755 // Floating-point truncation and stores need to be done separately.
756 setTruncStoreAction(MVT::f64, MVT::f32, Expand);
757 setTruncStoreAction(MVT::f128, MVT::f32, Expand);
758 setTruncStoreAction(MVT::f128, MVT::f64, Expand);
759
760 // We have 64-bit FPR<->GPR moves, but need special handling for
761 // 32-bit forms.
762 if (!Subtarget.hasVector()) {
765 }
766
767 // VASTART and VACOPY need to deal with the SystemZ-specific varargs
768 // structure, but VAEND is a no-op.
772
773 if (Subtarget.isTargetzOS()) {
774 // Handle address space casts between mixed sized pointers.
777 }
778
780
781 // Codes for which we want to perform some z-specific combinations.
785 ISD::LOAD,
798 ISD::SRL,
799 ISD::SRA,
800 ISD::MUL,
801 ISD::SDIV,
802 ISD::UDIV,
803 ISD::SREM,
804 ISD::UREM,
807
808 // Handle intrinsics.
811
812 // We're not using SJLJ for exception handling, but they're implemented
813 // solely to support use of __builtin_setjmp / __builtin_longjmp.
816
817 // We want to use MVC in preference to even a single load/store pair.
818 MaxStoresPerMemcpy = Subtarget.hasVector() ? 2 : 0;
820
821 // Same with memmove.
822 MaxStoresPerMemmove = Subtarget.hasVector() ? 2 : 0;
824
825 // The main memset sequence is a byte store followed by an MVC.
826 // Two STC or MV..I stores win over that, but the kind of fused stores
827 // generated by target-independent code don't when the byte value is
828 // variable. E.g. "STC <reg>;MHI <reg>,257;STH <reg>" is not better
829 // than "STC;MVC". Handle the choice in target-specific code instead.
830 MaxStoresPerMemset = Subtarget.hasVector() ? 2 : 0;
832
833 // Default to having -disable-strictnode-mutation on
834 IsStrictFPEnabled = true;
835}
836
838 return Subtarget.hasSoftFloat();
839}
840
842 LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT,
843 unsigned &NumIntermediates, MVT &RegisterVT) const {
844 // Pass fp16 vectors in VR(s).
845 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16)) {
846 IntermediateVT = RegisterVT = MVT::v8f16;
847 return NumIntermediates =
849 }
851 Context, CC, VT, IntermediateVT, NumIntermediates, RegisterVT);
852}
853
856 EVT VT) const {
857 // 128-bit single-element vector types are passed like other vectors,
858 // not like their element type.
859 if (Subtarget.hasVector() && VT.isVector() && VT.getSizeInBits() == 128 &&
860 VT.getVectorNumElements() == 1)
861 return MVT::v16i8;
862 // Pass fp16 vectors in VR(s).
863 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16))
864 return MVT::v8f16;
865 return TargetLowering::getRegisterTypeForCallingConv(Context, CC, VT);
866}
867
869 LLVMContext &Context, CallingConv::ID CC, EVT VT) const {
870 // Pass fp16 vectors in VR(s).
871 if (Subtarget.hasVector() && VT.isVectorOf(MVT::f16))
873 return TargetLowering::getNumRegistersForCallingConv(Context, CC, VT);
874}
875
877 LLVMContext &, EVT VT) const {
878 if (!VT.isVector())
879 return MVT::i32;
881}
882
884 const MachineFunction &MF, EVT VT) const {
885 if (useSoftFloat())
886 return false;
887
888 VT = VT.getScalarType();
889
890 if (!VT.isSimple())
891 return false;
892
893 switch (VT.getSimpleVT().SimpleTy) {
894 case MVT::f32:
895 case MVT::f64:
896 return true;
897 case MVT::f128:
898 return Subtarget.hasVectorEnhancements1();
899 default:
900 break;
901 }
902
903 return false;
904}
905
906// Return true if the constant can be generated with a vector instruction,
907// such as VGM, VGMB or VREPI.
909 const SystemZSubtarget &Subtarget) {
910 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
911 if (!Subtarget.hasVector() ||
912 (isFP128 && !Subtarget.hasVectorEnhancements1()))
913 return false;
914
915 // Try using VECTOR GENERATE BYTE MASK. This is the architecturally-
916 // preferred way of creating all-zero and all-one vectors so give it
917 // priority over other methods below.
918 unsigned Mask = 0;
919 unsigned I = 0;
920 for (; I < SystemZ::VectorBytes; ++I) {
921 uint64_t Byte = IntBits.lshr(I * 8).trunc(8).getZExtValue();
922 if (Byte == 0xff)
923 Mask |= 1ULL << I;
924 else if (Byte != 0)
925 break;
926 }
927 if (I == SystemZ::VectorBytes) {
928 Opcode = SystemZISD::BYTE_MASK;
929 OpVals.push_back(Mask);
931 return true;
932 }
933
934 if (SplatBitSize > 64)
935 return false;
936
937 auto TryValue = [&](uint64_t Value) -> bool {
938 // Try VECTOR REPLICATE IMMEDIATE
939 int64_t SignedValue = SignExtend64(Value, SplatBitSize);
940 if (isInt<16>(SignedValue)) {
941 OpVals.push_back(((unsigned) SignedValue));
942 Opcode = SystemZISD::REPLICATE;
944 SystemZ::VectorBits / SplatBitSize);
945 return true;
946 }
947 // Try VECTOR GENERATE MASK
948 unsigned Start, End;
949 if (TII->isRxSBGMask(Value, SplatBitSize, Start, End)) {
950 // isRxSBGMask returns the bit numbers for a full 64-bit value, with 0
951 // denoting 1 << 63 and 63 denoting 1. Convert them to bit numbers for
952 // an SplatBitSize value, so that 0 denotes 1 << (SplatBitSize-1).
953 OpVals.push_back(Start - (64 - SplatBitSize));
954 OpVals.push_back(End - (64 - SplatBitSize));
955 Opcode = SystemZISD::ROTATE_MASK;
957 SystemZ::VectorBits / SplatBitSize);
958 return true;
959 }
960 return false;
961 };
962
963 // First try assuming that any undefined bits above the highest set bit
964 // and below the lowest set bit are 1s. This increases the likelihood of
965 // being able to use a sign-extended element value in VECTOR REPLICATE
966 // IMMEDIATE or a wraparound mask in VECTOR GENERATE MASK.
967 uint64_t SplatBitsZ = SplatBits.getZExtValue();
968 uint64_t SplatUndefZ = SplatUndef.getZExtValue();
969 unsigned LowerBits = llvm::countr_zero(SplatBitsZ);
970 unsigned UpperBits = llvm::countl_zero(SplatBitsZ);
971 uint64_t Lower = SplatUndefZ & maskTrailingOnes<uint64_t>(LowerBits);
972 uint64_t Upper = SplatUndefZ & maskLeadingOnes<uint64_t>(UpperBits);
973 if (TryValue(SplatBitsZ | Upper | Lower))
974 return true;
975
976 // Now try assuming that any undefined bits between the first and
977 // last defined set bits are set. This increases the chances of
978 // using a non-wraparound mask.
979 uint64_t Middle = SplatUndefZ & ~Upper & ~Lower;
980 return TryValue(SplatBitsZ | Middle);
981}
982
984 if (IntImm.isSingleWord()) {
985 IntBits = APInt(128, IntImm.getZExtValue());
986 IntBits <<= (SystemZ::VectorBits - IntImm.getBitWidth());
987 } else
988 IntBits = IntImm;
989 assert(IntBits.getBitWidth() == 128 && "Unsupported APInt.");
990
991 // Find the smallest splat.
992 SplatBits = IntImm;
993 unsigned Width = SplatBits.getBitWidth();
994 while (Width > 8) {
995 unsigned HalfSize = Width / 2;
996 APInt HighValue = SplatBits.lshr(HalfSize).trunc(HalfSize);
997 APInt LowValue = SplatBits.trunc(HalfSize);
998
999 // If the two halves do not match, stop here.
1000 if (HighValue != LowValue || 8 > HalfSize)
1001 break;
1002
1003 SplatBits = HighValue;
1004 Width = HalfSize;
1005 }
1006 SplatUndef = 0;
1007 SplatBitSize = Width;
1008}
1009
1011 assert(BVN->isConstant() && "Expected a constant BUILD_VECTOR");
1012 bool HasAnyUndefs;
1013
1014 // Get IntBits by finding the 128 bit splat.
1015 BVN->isConstantSplat(IntBits, SplatUndef, SplatBitSize, HasAnyUndefs, 128,
1016 true);
1017
1018 // Get SplatBits by finding the 8 bit or greater splat.
1019 BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs, 8,
1020 true);
1021}
1022
1024 bool ForCodeSize) const {
1025 // We can load zero using LZ?R and negative zero using LZ?R;LC?BR.
1026 if (Imm.isZero() || Imm.isNegZero())
1027 return true;
1028
1030}
1031
1034 MachineBasicBlock *MBB) const {
1035 DebugLoc DL = MI.getDebugLoc();
1036 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
1037 const SystemZRegisterInfo *TRI = Subtarget.getRegisterInfo();
1038
1039 MachineFunction *MF = MBB->getParent();
1040 MachineRegisterInfo &MRI = MF->getRegInfo();
1041
1042 const BasicBlock *BB = MBB->getBasicBlock();
1043 MachineFunction::iterator I = ++MBB->getIterator();
1044
1045 Register DstReg = MI.getOperand(0).getReg();
1046 const TargetRegisterClass *RC = MRI.getRegClass(DstReg);
1047 assert(TRI->isTypeLegalForClass(*RC, MVT::i32) && "Invalid destination!");
1048 (void)TRI;
1049 Register MainDstReg = MRI.createVirtualRegister(RC);
1050 Register RestoreDstReg = MRI.createVirtualRegister(RC);
1051
1052 MVT PVT = getPointerTy(MF->getDataLayout());
1053 assert((PVT == MVT::i64 || PVT == MVT::i32) && "Invalid Pointer Size!");
1054 // For v = setjmp(buf), we generate.
1055 // Algorithm:
1056 //
1057 // ---------
1058 // | thisMBB |
1059 // ---------
1060 // |
1061 // ------------------------
1062 // | |
1063 // ---------- ---------------
1064 // | mainMBB | | restoreMBB |
1065 // | v = 0 | | v = 1 |
1066 // ---------- ---------------
1067 // | |
1068 // -------------------------
1069 // |
1070 // -----------------------------
1071 // | sinkMBB |
1072 // | phi(v_mainMBB,v_restoreMBB) |
1073 // -----------------------------
1074 // thisMBB:
1075 // buf[FPOffset] = Frame Pointer if hasFP.
1076 // buf[LabelOffset] = restoreMBB <-- takes address of restoreMBB.
1077 // buf[BCOffset] = Backchain value if building with -mbackchain.
1078 // buf[SPOffset] = Stack Pointer.
1079 // buf[LPOffset] = We never write this slot with R13, gcc stores R13 always.
1080 // SjLjSetup restoreMBB
1081 // mainMBB:
1082 // v_main = 0
1083 // sinkMBB:
1084 // v = phi(v_main, v_restore)
1085 // restoreMBB:
1086 // v_restore = 1
1087
1088 MachineBasicBlock *ThisMBB = MBB;
1089 MachineBasicBlock *MainMBB = MF->CreateMachineBasicBlock(BB);
1090 MachineBasicBlock *SinkMBB = MF->CreateMachineBasicBlock(BB);
1091 MachineBasicBlock *RestoreMBB = MF->CreateMachineBasicBlock(BB);
1092
1093 MF->insert(I, MainMBB);
1094 MF->insert(I, SinkMBB);
1095 MF->push_back(RestoreMBB);
1096 RestoreMBB->setMachineBlockAddressTaken();
1097
1099
1100 // Transfer the remainder of BB and its successor edges to sinkMBB.
1101 SinkMBB->splice(SinkMBB->begin(), MBB,
1102 std::next(MachineBasicBlock::iterator(MI)), MBB->end());
1104
1105 // thisMBB:
1106 const int64_t FPOffset = 0; // Slot 1.
1107 const int64_t LabelOffset = 1 * PVT.getStoreSize(); // Slot 2.
1108 const int64_t BCOffset = 2 * PVT.getStoreSize(); // Slot 3.
1109 const int64_t SPOffset = 3 * PVT.getStoreSize(); // Slot 4.
1110
1111 // Buf address.
1112 Register BufReg = MI.getOperand(1).getReg();
1113
1114 const TargetRegisterClass *PtrRC = getRegClassFor(PVT);
1115 Register LabelReg = MRI.createVirtualRegister(PtrRC);
1116
1117 // Prepare IP for longjmp.
1118 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::LARL), LabelReg)
1119 .addMBB(RestoreMBB);
1120 // Store IP for return from jmp, slot 2, offset = 1.
1121 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1122 .addReg(LabelReg)
1123 .addReg(BufReg)
1124 .addImm(LabelOffset)
1125 .addReg(0);
1126
1127 auto *SpecialRegs = Subtarget.getSpecialRegisters();
1128 bool HasFP = Subtarget.getFrameLowering()->hasFP(*MF);
1129 if (HasFP) {
1130 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1131 .addReg(SpecialRegs->getFramePointerRegister())
1132 .addReg(BufReg)
1133 .addImm(FPOffset)
1134 .addReg(0);
1135 }
1136
1137 // Store SP.
1138 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1139 .addReg(SpecialRegs->getStackPointerRegister())
1140 .addReg(BufReg)
1141 .addImm(SPOffset)
1142 .addReg(0);
1143
1144 // Slot 3(Offset = 2) Backchain value (if building with -mbackchain).
1145 bool BackChain = MF->getSubtarget<SystemZSubtarget>().hasBackChain();
1146 if (BackChain) {
1147 Register BCReg = MRI.createVirtualRegister(PtrRC);
1148 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
1149 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::LG), BCReg)
1150 .addReg(SpecialRegs->getStackPointerRegister())
1151 .addImm(TFL->getBackchainOffset(*MF))
1152 .addReg(0);
1153
1154 BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::STG))
1155 .addReg(BCReg)
1156 .addReg(BufReg)
1157 .addImm(BCOffset)
1158 .addReg(0);
1159 }
1160
1161 // Setup.
1162 MIB = BuildMI(*ThisMBB, MI, DL, TII->get(SystemZ::EH_SjLj_Setup))
1163 .addMBB(RestoreMBB);
1164
1165 const SystemZRegisterInfo *RegInfo = Subtarget.getRegisterInfo();
1166 MIB.addRegMask(RegInfo->getNoPreservedMask());
1167
1168 ThisMBB->addSuccessor(MainMBB);
1169 ThisMBB->addSuccessor(RestoreMBB);
1170
1171 // mainMBB:
1172 BuildMI(MainMBB, DL, TII->get(SystemZ::LHI), MainDstReg).addImm(0);
1173 MainMBB->addSuccessor(SinkMBB);
1174
1175 // sinkMBB:
1176 BuildMI(*SinkMBB, SinkMBB->begin(), DL, TII->get(SystemZ::PHI), DstReg)
1177 .addReg(MainDstReg)
1178 .addMBB(MainMBB)
1179 .addReg(RestoreDstReg)
1180 .addMBB(RestoreMBB);
1181
1182 // restoreMBB.
1183 BuildMI(RestoreMBB, DL, TII->get(SystemZ::LHI), RestoreDstReg).addImm(1);
1184 BuildMI(RestoreMBB, DL, TII->get(SystemZ::J)).addMBB(SinkMBB);
1185 RestoreMBB->addSuccessor(SinkMBB);
1186
1187 MI.eraseFromParent();
1188
1189 return SinkMBB;
1190}
1191
1194 MachineBasicBlock *MBB) const {
1195
1196 DebugLoc DL = MI.getDebugLoc();
1197 const TargetInstrInfo *TII = Subtarget.getInstrInfo();
1198
1199 MachineFunction *MF = MBB->getParent();
1200 MachineRegisterInfo &MRI = MF->getRegInfo();
1201
1202 MVT PVT = getPointerTy(MF->getDataLayout());
1203 assert((PVT == MVT::i64 || PVT == MVT::i32) && "Invalid Pointer Size!");
1204 Register BufReg = MI.getOperand(0).getReg();
1205 const TargetRegisterClass *RC = MRI.getRegClass(BufReg);
1206 auto *SpecialRegs = Subtarget.getSpecialRegisters();
1207
1208 Register Tmp = MRI.createVirtualRegister(RC);
1209 Register BCReg = MRI.createVirtualRegister(RC);
1210
1212
1213 const int64_t FPOffset = 0;
1214 const int64_t LabelOffset = 1 * PVT.getStoreSize();
1215 const int64_t BCOffset = 2 * PVT.getStoreSize();
1216 const int64_t SPOffset = 3 * PVT.getStoreSize();
1217 const int64_t LPOffset = 4 * PVT.getStoreSize();
1218
1219 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), Tmp)
1220 .addReg(BufReg)
1221 .addImm(LabelOffset)
1222 .addReg(0);
1223
1224 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG),
1225 SpecialRegs->getFramePointerRegister())
1226 .addReg(BufReg)
1227 .addImm(FPOffset)
1228 .addReg(0);
1229
1230 // We are restoring R13 even though we never stored in setjmp from llvm,
1231 // as gcc always stores R13 in builtin_setjmp. We could have mixed code
1232 // gcc setjmp and llvm longjmp.
1233 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), SystemZ::R13D)
1234 .addReg(BufReg)
1235 .addImm(LPOffset)
1236 .addReg(0);
1237
1238 bool BackChain = MF->getSubtarget<SystemZSubtarget>().hasBackChain();
1239 if (BackChain) {
1240 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG), BCReg)
1241 .addReg(BufReg)
1242 .addImm(BCOffset)
1243 .addReg(0);
1244 }
1245
1246 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::LG),
1247 SpecialRegs->getStackPointerRegister())
1248 .addReg(BufReg)
1249 .addImm(SPOffset)
1250 .addReg(0);
1251
1252 if (BackChain) {
1253 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
1254 BuildMI(*MBB, MI, DL, TII->get(SystemZ::STG))
1255 .addReg(BCReg)
1256 .addReg(SpecialRegs->getStackPointerRegister())
1257 .addImm(TFL->getBackchainOffset(*MF))
1258 .addReg(0);
1259 }
1260
1261 MIB = BuildMI(*MBB, MI, DL, TII->get(SystemZ::BR)).addReg(Tmp);
1262
1263 MI.eraseFromParent();
1264 return MBB;
1265}
1266
1267/// Returns true if stack probing through inline assembly is requested.
1269 // If the function specifically requests inline stack probes, emit them.
1270 if (MF.getFunction().hasFnAttribute("probe-stack"))
1271 return MF.getFunction().getFnAttribute("probe-stack").getValueAsString() ==
1272 "inline-asm";
1273 return false;
1274}
1275
1280
1285
1288 const AtomicRMWInst *RMW) const {
1289 // Don't expand subword operations as they require special treatment.
1290 if (RMW->getType()->isIntegerTy(8) || RMW->getType()->isIntegerTy(16))
1292
1293 // Don't expand if there is a target instruction available.
1294 if (Subtarget.hasInterlockedAccess1() &&
1295 (RMW->getType()->isIntegerTy(32) || RMW->getType()->isIntegerTy(64)) &&
1302
1304}
1305
1307 // We can use CGFI or CLGFI.
1308 return isInt<32>(Imm) || isUInt<32>(Imm);
1309}
1310
1312 // We can use ALGFI or SLGFI.
1313 return isUInt<32>(Imm) || isUInt<32>(-Imm);
1314}
1315
1317 EVT VT, unsigned, Align, MachineMemOperand::Flags, unsigned *Fast) const {
1318 // Unaligned accesses should never be slower than the expanded version.
1319 // We check specifically for aligned accesses in the few cases where
1320 // they are required.
1321 if (Fast)
1322 *Fast = 1;
1323 return true;
1324}
1325
1327 EVT VT = Y.getValueType();
1328
1329 // We can use NC(G)RK for types in GPRs ...
1330 if (VT == MVT::i32 || VT == MVT::i64)
1331 return Subtarget.hasMiscellaneousExtensions3();
1332
1333 // ... or VNC for types in VRs.
1334 if (VT.isVector() || VT == MVT::i128)
1335 return Subtarget.hasVector();
1336
1337 return false;
1338}
1339
1340// Information about the addressing mode for a memory access.
1342 // True if a long displacement is supported.
1344
1345 // True if use of index register is supported.
1347
1348 AddressingMode(bool LongDispl, bool IdxReg) :
1349 LongDisplacement(LongDispl), IndexReg(IdxReg) {}
1350};
1351
1352// Return the desired addressing mode for a Load which has only one use (in
1353// the same block) which is a Store.
1355 Type *Ty) {
1356 // With vector support a Load->Store combination may be combined to either
1357 // an MVC or vector operations and it seems to work best to allow the
1358 // vector addressing mode.
1359 if (HasVector)
1360 return AddressingMode(false/*LongDispl*/, true/*IdxReg*/);
1361
1362 // Otherwise only the MVC case is special.
1363 bool MVC = Ty->isIntegerTy(8);
1364 return AddressingMode(!MVC/*LongDispl*/, !MVC/*IdxReg*/);
1365}
1366
1367// Return the addressing mode which seems most desirable given an LLVM
1368// Instruction pointer.
1369static AddressingMode
1372 switch (II->getIntrinsicID()) {
1373 default: break;
1374 case Intrinsic::memset:
1375 case Intrinsic::memmove:
1376 case Intrinsic::memcpy:
1377 return AddressingMode(false/*LongDispl*/, false/*IdxReg*/);
1378 }
1379 }
1380
1381 if (isa<LoadInst>(I) && I->hasOneUse()) {
1382 auto *SingleUser = cast<Instruction>(*I->user_begin());
1383 if (SingleUser->getParent() == I->getParent()) {
1384 if (isa<ICmpInst>(SingleUser)) {
1385 if (auto *C = dyn_cast<ConstantInt>(SingleUser->getOperand(1)))
1386 if (C->getBitWidth() <= 64 &&
1387 (isInt<16>(C->getSExtValue()) || isUInt<16>(C->getZExtValue())))
1388 // Comparison of memory with 16 bit signed / unsigned immediate
1389 return AddressingMode(false/*LongDispl*/, false/*IdxReg*/);
1390 } else if (isa<StoreInst>(SingleUser))
1391 // Load->Store
1392 return getLoadStoreAddrMode(HasVector, I->getType());
1393 }
1394 } else if (auto *StoreI = dyn_cast<StoreInst>(I)) {
1395 if (auto *LoadI = dyn_cast<LoadInst>(StoreI->getValueOperand()))
1396 if (LoadI->hasOneUse() && LoadI->getParent() == I->getParent())
1397 // Load->Store
1398 return getLoadStoreAddrMode(HasVector, LoadI->getType());
1399 }
1400
1401 if (HasVector && (isa<LoadInst>(I) || isa<StoreInst>(I))) {
1402
1403 // * Use LDE instead of LE/LEY for z13 to avoid partial register
1404 // dependencies (LDE only supports small offsets).
1405 // * Utilize the vector registers to hold floating point
1406 // values (vector load / store instructions only support small
1407 // offsets).
1408
1409 Type *MemAccessTy = (isa<LoadInst>(I) ? I->getType() :
1410 I->getOperand(0)->getType());
1411 bool IsFPAccess = MemAccessTy->isFloatingPointTy();
1412 bool IsVectorAccess = MemAccessTy->isVectorTy();
1413
1414 // A store of an extracted vector element will be combined into a VSTE type
1415 // instruction.
1416 if (!IsVectorAccess && isa<StoreInst>(I)) {
1417 Value *DataOp = I->getOperand(0);
1418 if (isa<ExtractElementInst>(DataOp))
1419 IsVectorAccess = true;
1420 }
1421
1422 // A load which gets inserted into a vector element will be combined into a
1423 // VLE type instruction.
1424 if (!IsVectorAccess && isa<LoadInst>(I) && I->hasOneUse()) {
1425 User *LoadUser = *I->user_begin();
1426 if (isa<InsertElementInst>(LoadUser))
1427 IsVectorAccess = true;
1428 }
1429
1430 if (IsFPAccess || IsVectorAccess)
1431 return AddressingMode(false/*LongDispl*/, true/*IdxReg*/);
1432 }
1433
1434 return AddressingMode(true/*LongDispl*/, true/*IdxReg*/);
1435}
1436
1438 const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I) const {
1439 // Punt on globals for now, although they can be used in limited
1440 // RELATIVE LONG cases.
1441 if (AM.BaseGV)
1442 return false;
1443
1444 // Require a 20-bit signed offset.
1445 if (!isInt<20>(AM.BaseOffs))
1446 return false;
1447
1448 bool RequireD12 =
1449 Subtarget.hasVector() && (Ty->isVectorTy() || Ty->isIntegerTy(128));
1450 AddressingMode SupportedAM(!RequireD12, true);
1451 if (I != nullptr)
1452 SupportedAM = supportedAddressingMode(I, Subtarget.hasVector());
1453
1454 if (!SupportedAM.LongDisplacement && !isUInt<12>(AM.BaseOffs))
1455 return false;
1456
1457 if (!SupportedAM.IndexReg)
1458 // No indexing allowed.
1459 return AM.Scale == 0;
1460 else
1461 // Indexing is OK but no scale factor can be applied.
1462 return AM.Scale == 0 || AM.Scale == 1;
1463}
1464
1466 LLVMContext &Context, std::vector<EVT> &MemOps, unsigned Limit,
1467 const MemOp &Op, unsigned DstAS, unsigned SrcAS,
1468 const AttributeList &FuncAttributes, EVT *LargestVT) const {
1469
1470 assert(Limit != ~0U &&
1471 "Expected EmitTargetCodeForMemXXX() to handle AlwaysInline cases.");
1472
1473 if (Op.isZeroMemset())
1474 return false; // Memset zero: Use XC.
1475
1476 const int MVCFastLen = 16;
1477 // Use MVC up to 16 bytes for memcpy. Small memset uses STC/MVI for first
1478 // byte.
1479 if (Op.isMemcpy() && Op.size() <= MVCFastLen)
1480 return false;
1481 if (Op.isMemset() && Op.size() - 1 <= MVCFastLen)
1482 return false;
1483
1484 // Avoid unaligned VL/VST:s.
1485 if ((Op.size() >= 16 && !Op.isAligned(Align(8))) ||
1486 (Op.size() >= 25 && Op.size() <= 31))
1487 return false;
1488
1490 Context, MemOps, Limit, Op, DstAS, SrcAS, FuncAttributes, LargestVT);
1491}
1492
1494 LLVMContext &Context, const MemOp &Op,
1495 const AttributeList &FuncAttributes) const {
1496 return Subtarget.hasVector() ? MVT::v2i64 : MVT::Other;
1497}
1498
1499bool SystemZTargetLowering::isTruncateFree(Type *FromType, Type *ToType) const {
1500 if (!FromType->isIntegerTy() || !ToType->isIntegerTy())
1501 return false;
1502 unsigned FromBits = FromType->getPrimitiveSizeInBits().getFixedValue();
1503 unsigned ToBits = ToType->getPrimitiveSizeInBits().getFixedValue();
1504 return FromBits > ToBits;
1505}
1506
1508 if (!FromVT.isInteger() || !ToVT.isInteger())
1509 return false;
1510 unsigned FromBits = FromVT.getFixedSizeInBits();
1511 unsigned ToBits = ToVT.getFixedSizeInBits();
1512 return FromBits > ToBits;
1513}
1514
1515//===----------------------------------------------------------------------===//
1516// Inline asm support
1517//===----------------------------------------------------------------------===//
1518
1521 if (Constraint.size() == 1) {
1522 switch (Constraint[0]) {
1523 case 'a': // Address register
1524 case 'd': // Data register (equivalent to 'r')
1525 case 'f': // Floating-point register
1526 case 'h': // High-part register
1527 case 'r': // General-purpose register
1528 case 'v': // Vector register
1529 return C_RegisterClass;
1530
1531 case 'Q': // Memory with base and unsigned 12-bit displacement
1532 case 'R': // Likewise, plus an index
1533 case 'S': // Memory with base and signed 20-bit displacement
1534 case 'T': // Likewise, plus an index
1535 case 'm': // Equivalent to 'T'.
1536 return C_Memory;
1537
1538 case 'I': // Unsigned 8-bit constant
1539 case 'J': // Unsigned 12-bit constant
1540 case 'K': // Signed 16-bit constant
1541 case 'L': // Signed 20-bit displacement (on all targets we support)
1542 case 'M': // 0x7fffffff
1543 return C_Immediate;
1544
1545 default:
1546 break;
1547 }
1548 } else if (Constraint.size() == 2 && Constraint[0] == 'Z') {
1549 switch (Constraint[1]) {
1550 case 'Q': // Address with base and unsigned 12-bit displacement
1551 case 'R': // Likewise, plus an index
1552 case 'S': // Address with base and signed 20-bit displacement
1553 case 'T': // Likewise, plus an index
1554 return C_Address;
1555
1556 default:
1557 break;
1558 }
1559 } else if (Constraint.size() == 5 && Constraint.starts_with("{")) {
1560 if (StringRef("{@cc}").compare(Constraint) == 0)
1561 return C_Other;
1562 }
1563 return TargetLowering::getConstraintType(Constraint);
1564}
1565
1568 AsmOperandInfo &Info, const char *Constraint) const {
1570 Value *CallOperandVal = Info.CallOperandVal;
1571 // If we don't have a value, we can't do a match,
1572 // but allow it at the lowest weight.
1573 if (!CallOperandVal)
1574 return CW_Default;
1575 Type *type = CallOperandVal->getType();
1576 // Look at the constraint type.
1577 switch (*Constraint) {
1578 default:
1579 Weight = TargetLowering::getSingleConstraintMatchWeight(Info, Constraint);
1580 break;
1581
1582 case 'a': // Address register
1583 case 'd': // Data register (equivalent to 'r')
1584 case 'h': // High-part register
1585 case 'r': // General-purpose register
1586 Weight =
1587 CallOperandVal->getType()->isIntegerTy() ? CW_Register : CW_Default;
1588 break;
1589
1590 case 'f': // Floating-point register
1591 if (!useSoftFloat())
1592 Weight = type->isFloatingPointTy() ? CW_Register : CW_Default;
1593 break;
1594
1595 case 'v': // Vector register
1596 if (Subtarget.hasVector())
1597 Weight = (type->isVectorTy() || type->isFloatingPointTy()) ? CW_Register
1598 : CW_Default;
1599 break;
1600
1601 case 'I': // Unsigned 8-bit constant
1602 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1603 if (isUInt<8>(C->getZExtValue()))
1604 Weight = CW_Constant;
1605 break;
1606
1607 case 'J': // Unsigned 12-bit constant
1608 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1609 if (isUInt<12>(C->getZExtValue()))
1610 Weight = CW_Constant;
1611 break;
1612
1613 case 'K': // Signed 16-bit constant
1614 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1615 if (isInt<16>(C->getSExtValue()))
1616 Weight = CW_Constant;
1617 break;
1618
1619 case 'L': // Signed 20-bit displacement (on all targets we support)
1620 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1621 if (isInt<20>(C->getSExtValue()))
1622 Weight = CW_Constant;
1623 break;
1624
1625 case 'M': // 0x7fffffff
1626 if (auto *C = dyn_cast<ConstantInt>(CallOperandVal))
1627 if (C->getZExtValue() == 0x7fffffff)
1628 Weight = CW_Constant;
1629 break;
1630 }
1631 return Weight;
1632}
1633
1634// Parse a "{tNNN}" register constraint for which the register type "t"
1635// has already been verified. MC is the class associated with "t" and
1636// Map maps 0-based register numbers to LLVM register numbers.
1637static std::pair<unsigned, const TargetRegisterClass *>
1639 const unsigned *Map, unsigned Size) {
1640 assert(*(Constraint.end()-1) == '}' && "Missing '}'");
1641 if (isdigit(Constraint[2])) {
1642 unsigned Index;
1643 bool Failed =
1644 Constraint.slice(2, Constraint.size() - 1).getAsInteger(10, Index);
1645 if (!Failed && Index < Size && Map[Index])
1646 return std::make_pair(Map[Index], RC);
1647 }
1648 return std::make_pair(0U, nullptr);
1649}
1650
1651std::pair<unsigned, const TargetRegisterClass *>
1653 const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const {
1654 if (Constraint.size() == 1) {
1655 // GCC Constraint Letters
1656 switch (Constraint[0]) {
1657 default: break;
1658 case 'd': // Data register (equivalent to 'r')
1659 case 'r': // General-purpose register
1660 if (VT.getSizeInBits() == 64)
1661 return std::make_pair(0U, &SystemZ::GR64BitRegClass);
1662 else if (VT.getSizeInBits() == 128)
1663 return std::make_pair(0U, &SystemZ::GR128BitRegClass);
1664 return std::make_pair(0U, &SystemZ::GR32BitRegClass);
1665
1666 case 'a': // Address register
1667 if (VT == MVT::i64)
1668 return std::make_pair(0U, &SystemZ::ADDR64BitRegClass);
1669 else if (VT == MVT::i128)
1670 return std::make_pair(0U, &SystemZ::ADDR128BitRegClass);
1671 return std::make_pair(0U, &SystemZ::ADDR32BitRegClass);
1672
1673 case 'h': // High-part register (an LLVM extension)
1674 return std::make_pair(0U, &SystemZ::GRH32BitRegClass);
1675
1676 case 'f': // Floating-point register
1677 if (!useSoftFloat()) {
1678 if (VT.getSizeInBits() == 16)
1679 return std::make_pair(0U, &SystemZ::FP16BitRegClass);
1680 else if (VT.getSizeInBits() == 64)
1681 return std::make_pair(0U, &SystemZ::FP64BitRegClass);
1682 else if (VT.getSizeInBits() == 128)
1683 return std::make_pair(0U, &SystemZ::FP128BitRegClass);
1684 return std::make_pair(0U, &SystemZ::FP32BitRegClass);
1685 }
1686 break;
1687
1688 case 'v': // Vector register
1689 if (Subtarget.hasVector()) {
1690 if (VT.getSizeInBits() == 16)
1691 return std::make_pair(0U, &SystemZ::VR16BitRegClass);
1692 if (VT.getSizeInBits() == 32)
1693 return std::make_pair(0U, &SystemZ::VR32BitRegClass);
1694 if (VT.getSizeInBits() == 64)
1695 return std::make_pair(0U, &SystemZ::VR64BitRegClass);
1696 return std::make_pair(0U, &SystemZ::VR128BitRegClass);
1697 }
1698 break;
1699 }
1700 }
1701 if (Constraint.starts_with("{")) {
1702
1703 // A clobber constraint (e.g. ~{f0}) will have MVT::Other which is illegal
1704 // to check the size on.
1705 auto getVTSizeInBits = [&VT]() {
1706 return VT == MVT::Other ? 0 : VT.getSizeInBits();
1707 };
1708
1709 // We need to override the default register parsing for GPRs and FPRs
1710 // because the interpretation depends on VT. The internal names of
1711 // the registers are also different from the external names
1712 // (F0D and F0S instead of F0, etc.).
1713 if (Constraint[1] == 'r') {
1714 if (getVTSizeInBits() == 32)
1715 return parseRegisterNumber(Constraint, &SystemZ::GR32BitRegClass,
1717 if (getVTSizeInBits() == 128)
1718 return parseRegisterNumber(Constraint, &SystemZ::GR128BitRegClass,
1720 return parseRegisterNumber(Constraint, &SystemZ::GR64BitRegClass,
1722 }
1723 if (Constraint[1] == 'f') {
1724 if (useSoftFloat())
1725 return std::make_pair(
1726 0u, static_cast<const TargetRegisterClass *>(nullptr));
1727 if (getVTSizeInBits() == 16)
1728 return parseRegisterNumber(Constraint, &SystemZ::FP16BitRegClass,
1730 if (getVTSizeInBits() == 32)
1731 return parseRegisterNumber(Constraint, &SystemZ::FP32BitRegClass,
1733 if (getVTSizeInBits() == 128)
1734 return parseRegisterNumber(Constraint, &SystemZ::FP128BitRegClass,
1736 return parseRegisterNumber(Constraint, &SystemZ::FP64BitRegClass,
1738 }
1739 if (Constraint[1] == 'v') {
1740 if (!Subtarget.hasVector())
1741 return std::make_pair(
1742 0u, static_cast<const TargetRegisterClass *>(nullptr));
1743 if (getVTSizeInBits() == 16)
1744 return parseRegisterNumber(Constraint, &SystemZ::VR16BitRegClass,
1746 if (getVTSizeInBits() == 32)
1747 return parseRegisterNumber(Constraint, &SystemZ::VR32BitRegClass,
1749 if (getVTSizeInBits() == 64)
1750 return parseRegisterNumber(Constraint, &SystemZ::VR64BitRegClass,
1752 return parseRegisterNumber(Constraint, &SystemZ::VR128BitRegClass,
1754 }
1755 if (Constraint[1] == '@') {
1756 if (StringRef("{@cc}").compare(Constraint) == 0)
1757 return std::make_pair(SystemZ::CC, &SystemZ::CCRRegClass);
1758 }
1759 }
1760 return TargetLowering::getRegForInlineAsmConstraint(TRI, Constraint, VT);
1761}
1762
1763// FIXME? Maybe this could be a TableGen attribute on some registers and
1764// this table could be generated automatically from RegInfo.
1767 const MachineFunction &MF) const {
1768 Register Reg =
1770 .Case("r4", Subtarget.isTargetXPLINK64() ? SystemZ::R4D
1771 : SystemZ::NoRegister)
1772 .Case("r15",
1773 Subtarget.isTargetELF() ? SystemZ::R15D : SystemZ::NoRegister)
1774 .Default(Register());
1775
1776 return Reg;
1777}
1778
1780 ExceptionHandling EH, const Constant *PersonalityFn) const {
1781 return Subtarget.isTargetXPLINK64() ? SystemZ::R1D : SystemZ::R6D;
1782}
1783
1785 ExceptionHandling EH, const Constant *PersonalityFn) const {
1786 return Subtarget.isTargetXPLINK64() ? SystemZ::R2D : SystemZ::R7D;
1787}
1788
1789// Convert condition code in CCReg to an i32 value.
1791 SDLoc DL(CCReg);
1792 SDValue IPM = DAG.getNode(SystemZISD::IPM, DL, MVT::i32, CCReg);
1793 return DAG.getNode(ISD::SRL, DL, MVT::i32, IPM,
1794 DAG.getConstant(SystemZ::IPM_CC, DL, MVT::i32));
1795}
1796
1797// Lower @cc targets via setcc.
1799 SDValue &Chain, SDValue &Glue, const SDLoc &DL,
1800 const AsmOperandInfo &OpInfo, SelectionDAG &DAG) const {
1801 if (StringRef("{@cc}").compare(OpInfo.ConstraintCode) != 0)
1802 return SDValue();
1803
1804 // Check that return type is valid.
1805 if (OpInfo.ConstraintVT.isVector() || !OpInfo.ConstraintVT.isInteger() ||
1806 OpInfo.ConstraintVT.getSizeInBits() < 8)
1807 report_fatal_error("Glue output operand is of invalid type");
1808
1809 if (Glue.getNode()) {
1810 Glue = DAG.getCopyFromReg(Chain, DL, SystemZ::CC, MVT::i32, Glue);
1811 Chain = Glue.getValue(1);
1812 } else
1813 Glue = DAG.getCopyFromReg(Chain, DL, SystemZ::CC, MVT::i32);
1814 return getCCResult(DAG, Glue);
1815}
1816
1818 SDValue Op, StringRef Constraint, std::vector<SDValue> &Ops,
1819 SelectionDAG &DAG) const {
1820 // Only support length 1 constraints for now.
1821 if (Constraint.size() == 1) {
1822 switch (Constraint[0]) {
1823 case 'I': // Unsigned 8-bit constant
1824 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1825 if (isUInt<8>(C->getZExtValue()))
1826 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1827 Op.getValueType()));
1828 return;
1829
1830 case 'J': // Unsigned 12-bit constant
1831 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1832 if (isUInt<12>(C->getZExtValue()))
1833 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1834 Op.getValueType()));
1835 return;
1836
1837 case 'K': // Signed 16-bit constant
1838 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1839 if (isInt<16>(C->getSExtValue()))
1840 Ops.push_back(DAG.getSignedTargetConstant(
1841 C->getSExtValue(), SDLoc(Op), Op.getValueType()));
1842 return;
1843
1844 case 'L': // Signed 20-bit displacement (on all targets we support)
1845 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1846 if (isInt<20>(C->getSExtValue()))
1847 Ops.push_back(DAG.getSignedTargetConstant(
1848 C->getSExtValue(), SDLoc(Op), Op.getValueType()));
1849 return;
1850
1851 case 'M': // 0x7fffffff
1852 if (auto *C = dyn_cast<ConstantSDNode>(Op))
1853 if (C->getZExtValue() == 0x7fffffff)
1854 Ops.push_back(DAG.getTargetConstant(C->getZExtValue(), SDLoc(Op),
1855 Op.getValueType()));
1856 return;
1857 }
1858 }
1860}
1861
1862//===----------------------------------------------------------------------===//
1863// Calling conventions
1864//===----------------------------------------------------------------------===//
1865
1866#define GET_CALLING_CONV_IMPL
1867#include "SystemZGenCallingConv.inc"
1868
1870 CallingConv::ID) const {
1871 static const MCPhysReg ScratchRegs[] = { SystemZ::R0D, SystemZ::R1D,
1872 SystemZ::R14D, 0 };
1873 return ScratchRegs;
1874}
1875
1877 Type *ToType) const {
1878 return isTruncateFree(FromType, ToType);
1879}
1880
1882 return CI->isTailCall();
1883}
1884
1885// Value is a value that has been passed to us in the location described by VA
1886// (and so has type VA.getLocVT()). Convert Value to VA.getValVT(), chaining
1887// any loads onto Chain.
1889 CCValAssign &VA, SDValue Chain,
1890 SDValue Value) {
1891 // If the argument has been promoted from a smaller type, insert an
1892 // assertion to capture this.
1893 if (VA.getLocInfo() == CCValAssign::SExt)
1895 DAG.getValueType(VA.getValVT()));
1896 else if (VA.getLocInfo() == CCValAssign::ZExt)
1898 DAG.getValueType(VA.getValVT()));
1899
1900 if (VA.isExtInLoc())
1901 Value = DAG.getNode(ISD::TRUNCATE, DL, VA.getValVT(), Value);
1902 else if (VA.getLocInfo() == CCValAssign::BCvt) {
1903 // If this is a short vector argument loaded from the stack,
1904 // extend from i64 to full vector size and then bitcast.
1905 assert(VA.getLocVT() == MVT::i64);
1906 assert(VA.getValVT().isVector());
1907 Value = DAG.getBuildVector(MVT::v2i64, DL, {Value, DAG.getUNDEF(MVT::i64)});
1908 Value = DAG.getNode(ISD::BITCAST, DL, VA.getValVT(), Value);
1909 } else
1910 assert(VA.getLocInfo() == CCValAssign::Full && "Unsupported getLocInfo");
1911 return Value;
1912}
1913
1914// Value is a value of type VA.getValVT() that we need to copy into
1915// the location described by VA. Return a copy of Value converted to
1916// VA.getValVT(). The caller is responsible for handling indirect values.
1918 CCValAssign &VA, SDValue Value) {
1919 switch (VA.getLocInfo()) {
1920 case CCValAssign::SExt:
1921 return DAG.getNode(ISD::SIGN_EXTEND, DL, VA.getLocVT(), Value);
1922 case CCValAssign::ZExt:
1923 return DAG.getNode(ISD::ZERO_EXTEND, DL, VA.getLocVT(), Value);
1924 case CCValAssign::AExt:
1925 return DAG.getNode(ISD::ANY_EXTEND, DL, VA.getLocVT(), Value);
1926 case CCValAssign::BCvt: {
1927 assert(VA.getLocVT() == MVT::i64 || VA.getLocVT() == MVT::i128);
1928 assert(VA.getValVT().isVector() || VA.getValVT() == MVT::f32 ||
1929 VA.getValVT() == MVT::f64 || VA.getValVT() == MVT::f128);
1930 // For an f32 vararg we need to first promote it to an f64 and then
1931 // bitcast it to an i64.
1932 if (VA.getValVT() == MVT::f32 && VA.getLocVT() == MVT::i64)
1933 Value = DAG.getNode(ISD::FP_EXTEND, DL, MVT::f64, Value);
1934 MVT BitCastToType = VA.getValVT().isVector() && VA.getLocVT() == MVT::i64
1935 ? MVT::v2i64
1936 : VA.getLocVT();
1937 Value = DAG.getNode(ISD::BITCAST, DL, BitCastToType, Value);
1938 // For ELF, this is a short vector argument to be stored to the stack,
1939 // bitcast to v2i64 and then extract first element.
1940 if (BitCastToType == MVT::v2i64)
1941 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, VA.getLocVT(), Value,
1942 DAG.getConstant(0, DL, MVT::i32));
1943 return Value;
1944 }
1945 case CCValAssign::Full:
1946 return Value;
1947 default:
1948 llvm_unreachable("Unhandled getLocInfo()");
1949 }
1950}
1951
1953 SDLoc DL(In);
1954 SDValue Lo, Hi;
1955 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128)) {
1956 Lo = DAG.getNode(ISD::TRUNCATE, DL, MVT::i64, In);
1957 Hi = DAG.getNode(ISD::TRUNCATE, DL, MVT::i64,
1958 DAG.getNode(ISD::SRL, DL, MVT::i128, In,
1959 DAG.getConstant(64, DL, MVT::i32)));
1960 } else {
1961 std::tie(Lo, Hi) = DAG.SplitScalar(In, DL, MVT::i64, MVT::i64);
1962 }
1963
1964 // FIXME: If v2i64 were a legal type, we could use it instead of
1965 // Untyped here. This might enable improved folding.
1966 SDNode *Pair = DAG.getMachineNode(SystemZ::PAIR128, DL,
1967 MVT::Untyped, Hi, Lo);
1968 return SDValue(Pair, 0);
1969}
1970
1972 SDLoc DL(In);
1973 SDValue Hi = DAG.getTargetExtractSubreg(SystemZ::subreg_h64,
1974 DL, MVT::i64, In);
1975 SDValue Lo = DAG.getTargetExtractSubreg(SystemZ::subreg_l64,
1976 DL, MVT::i64, In);
1977
1978 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128)) {
1979 Lo = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i128, Lo);
1980 Hi = DAG.getNode(ISD::ZERO_EXTEND, DL, MVT::i128, Hi);
1981 Hi = DAG.getNode(ISD::SHL, DL, MVT::i128, Hi,
1982 DAG.getConstant(64, DL, MVT::i32));
1983 return DAG.getNode(ISD::OR, DL, MVT::i128, Lo, Hi);
1984 } else {
1985 return DAG.getNode(ISD::BUILD_PAIR, DL, MVT::i128, Lo, Hi);
1986 }
1987}
1988
1990 SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts,
1991 unsigned NumParts, MVT PartVT, std::optional<CallingConv::ID> CC) const {
1992 EVT ValueVT = Val.getValueType();
1993 if (ValueVT.getSizeInBits() == 128 && NumParts == 1 && PartVT == MVT::Untyped) {
1994 // Inline assembly operand.
1995 Parts[0] = lowerI128ToGR128(DAG, DAG.getBitcast(MVT::i128, Val));
1996 return true;
1997 }
1998
1999 return false;
2000}
2001
2003 SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts,
2004 MVT PartVT, EVT ValueVT, std::optional<CallingConv::ID> CC) const {
2005 if (ValueVT.getSizeInBits() == 128 && NumParts == 1 && PartVT == MVT::Untyped) {
2006 // Inline assembly operand.
2007 SDValue Res = lowerGR128ToI128(DAG, Parts[0]);
2008 return DAG.getBitcast(ValueVT, Res);
2009 }
2010
2011 return SDValue();
2012}
2013
2014// The first part of a split stack argument is at index I in Args (and
2015// ArgLocs). Return the type of a part and the number of them by reference.
2016template <class ArgTy>
2018 SmallVector<CCValAssign, 16> &ArgLocs, unsigned I,
2019 MVT &PartVT, unsigned &NumParts) {
2020 if (!Args[I].Flags.isSplit())
2021 return false;
2022 assert(I < ArgLocs.size() && ArgLocs.size() == Args.size() &&
2023 "ArgLocs havoc.");
2024 PartVT = ArgLocs[I].getValVT();
2025 NumParts = 1;
2026 for (unsigned PartIdx = I + 1;; ++PartIdx) {
2027 assert(PartIdx != ArgLocs.size() && "SplitEnd not found.");
2028 assert(ArgLocs[PartIdx].getValVT() == PartVT && "Unsupported split.");
2029 ++NumParts;
2030 if (Args[PartIdx].Flags.isSplitEnd())
2031 break;
2032 }
2033 return true;
2034}
2035
2037 SDValue Chain, CallingConv::ID CallConv, bool IsVarArg,
2038 const SmallVectorImpl<ISD::InputArg> &Ins, const SDLoc &DL,
2039 SelectionDAG &DAG, SmallVectorImpl<SDValue> &InVals) const {
2041 MachineFrameInfo &MFI = MF.getFrameInfo();
2042 MachineRegisterInfo &MRI = MF.getRegInfo();
2043 SystemZMachineFunctionInfo *FuncInfo =
2045 auto *TFL = Subtarget.getFrameLowering<SystemZELFFrameLowering>();
2046 EVT PtrVT = getPointerTy(DAG.getDataLayout());
2047
2048 // Assign locations to all of the incoming arguments.
2050 CCState CCInfo(CallConv, IsVarArg, MF, ArgLocs, *DAG.getContext());
2051 CCInfo.AnalyzeFormalArguments(Ins, CC_SystemZ);
2052 FuncInfo->setSizeOfFnParams(CCInfo.getStackSize());
2053
2054 unsigned NumFixedGPRs = 0;
2055 unsigned NumFixedFPRs = 0;
2056 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2057 SDValue ArgValue;
2058 CCValAssign &VA = ArgLocs[I];
2059 EVT LocVT = VA.getLocVT();
2060 if (VA.isRegLoc()) {
2061 // Arguments passed in registers
2062 const TargetRegisterClass *RC;
2063 switch (LocVT.getSimpleVT().SimpleTy) {
2064 default:
2065 // Integers smaller than i64 should be promoted to i64.
2066 llvm_unreachable("Unexpected argument type");
2067 case MVT::i32:
2068 NumFixedGPRs += 1;
2069 RC = &SystemZ::GR32BitRegClass;
2070 break;
2071 case MVT::i64:
2072 NumFixedGPRs += 1;
2073 RC = &SystemZ::GR64BitRegClass;
2074 break;
2075 case MVT::f16:
2076 NumFixedFPRs += 1;
2077 RC = &SystemZ::FP16BitRegClass;
2078 break;
2079 case MVT::f32:
2080 NumFixedFPRs += 1;
2081 RC = &SystemZ::FP32BitRegClass;
2082 break;
2083 case MVT::f64:
2084 NumFixedFPRs += 1;
2085 RC = &SystemZ::FP64BitRegClass;
2086 break;
2087 case MVT::f128:
2088 NumFixedFPRs += 2;
2089 RC = &SystemZ::FP128BitRegClass;
2090 break;
2091 case MVT::v16i8:
2092 case MVT::v8i16:
2093 case MVT::v4i32:
2094 case MVT::v2i64:
2095 case MVT::v8f16:
2096 case MVT::v4f32:
2097 case MVT::v2f64:
2098 RC = &SystemZ::VR128BitRegClass;
2099 break;
2100 }
2101
2102 Register VReg = MRI.createVirtualRegister(RC);
2103 MRI.addLiveIn(VA.getLocReg(), VReg);
2104 ArgValue = DAG.getCopyFromReg(Chain, DL, VReg, LocVT);
2105 } else {
2106 assert(VA.isMemLoc() && "Argument not register or memory");
2107
2108 // Create the frame index object for this incoming parameter.
2109 // FIXME: Pre-include call frame size in the offset, should not
2110 // need to manually add it here.
2111 int64_t ArgSPOffset = VA.getLocMemOffset();
2112 if (Subtarget.isTargetXPLINK64()) {
2113 auto &XPRegs =
2114 Subtarget.getSpecialRegisters<SystemZXPLINK64Registers>();
2115 ArgSPOffset += XPRegs.getCallFrameSize();
2116 }
2117 int FI =
2118 MFI.CreateFixedObject(LocVT.getSizeInBits() / 8, ArgSPOffset, true);
2119
2120 // Create the SelectionDAG nodes corresponding to a load
2121 // from this parameter. Unpromoted ints and floats are
2122 // passed as right-justified 8-byte values.
2123 SDValue FIN = DAG.getFrameIndex(FI, PtrVT);
2124 if (VA.getLocVT() == MVT::i32 || VA.getLocVT() == MVT::f32 ||
2125 VA.getLocVT() == MVT::f16) {
2126 unsigned SlotOffs = VA.getLocVT() == MVT::f16 ? 6 : 4;
2127 FIN = DAG.getNode(ISD::ADD, DL, PtrVT, FIN,
2128 DAG.getIntPtrConstant(SlotOffs, DL));
2129 }
2130 ArgValue = DAG.getLoad(LocVT, DL, Chain, FIN,
2132 }
2133
2134 // Convert the value of the argument register into the value that's
2135 // being passed.
2136 if (VA.getLocInfo() == CCValAssign::Indirect) {
2137 InVals.push_back(DAG.getLoad(VA.getValVT(), DL, Chain, ArgValue,
2139 // If the original argument was split (e.g. i128), we need
2140 // to load all parts of it here (using the same address).
2141 MVT PartVT;
2142 unsigned NumParts;
2143 if (analyzeArgSplit(Ins, ArgLocs, I, PartVT, NumParts)) {
2144 for (unsigned PartIdx = 1; PartIdx < NumParts; ++PartIdx) {
2145 ++I;
2146 CCValAssign &PartVA = ArgLocs[I];
2147 unsigned PartOffset = Ins[I].PartOffset;
2148 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, ArgValue,
2149 DAG.getIntPtrConstant(PartOffset, DL));
2150 InVals.push_back(DAG.getLoad(PartVA.getValVT(), DL, Chain, Address,
2152 assert(PartOffset && "Offset should be non-zero.");
2153 }
2154 }
2155 } else if (Subtarget.isTargetXPLINK64() &&
2156 (VA.getLocInfo() == CCValAssign::SExt ||
2157 VA.getLocInfo() == CCValAssign::ZExt) &&
2158 Ins[I].ArgVT.isSimple()) {
2159 // Some prior z/OS compilers do not always perform the extension of
2160 // short integer arguments or pointers. To accommodate those, do not
2161 // rely on that extension by avoiding any AssertSext/AssertZext nodes by
2162 // directly truncating ArgValue to the original argument type.
2163 MVT OrigVT = Ins[I].ArgVT.getSimpleVT();
2164 InVals.push_back(DAG.getNode(ISD::TRUNCATE, DL, OrigVT, ArgValue));
2165 } else
2166 InVals.push_back(convertLocVTToValVT(DAG, DL, VA, Chain, ArgValue));
2167 }
2168
2169 if (IsVarArg && Subtarget.isTargetXPLINK64()) {
2170 // Save the number of non-varargs registers for later use by va_start, etc.
2171 FuncInfo->setVarArgsFirstGPR(NumFixedGPRs);
2172 FuncInfo->setVarArgsFirstFPR(NumFixedFPRs);
2173
2174 auto *Regs = static_cast<SystemZXPLINK64Registers *>(
2175 Subtarget.getSpecialRegisters());
2176
2177 // Likewise the address (in the form of a frame index) of where the
2178 // first stack vararg would be. The 1-byte size here is arbitrary.
2179 // FIXME: Pre-include call frame size in the offset, should not
2180 // need to manually add it here.
2181 int64_t VarArgOffset = CCInfo.getStackSize() + Regs->getCallFrameSize();
2182 int FI = MFI.CreateFixedObject(1, VarArgOffset, true);
2183 FuncInfo->setVarArgsFrameIndex(FI);
2184 }
2185
2186 if (IsVarArg && Subtarget.isTargetELF()) {
2187 // Save the number of non-varargs registers for later use by va_start, etc.
2188 FuncInfo->setVarArgsFirstGPR(NumFixedGPRs);
2189 FuncInfo->setVarArgsFirstFPR(NumFixedFPRs);
2190
2191 // Likewise the address (in the form of a frame index) of where the
2192 // first stack vararg would be. The 1-byte size here is arbitrary.
2193 int64_t VarArgsOffset = CCInfo.getStackSize();
2194 FuncInfo->setVarArgsFrameIndex(
2195 MFI.CreateFixedObject(1, VarArgsOffset, true));
2196
2197 // ...and a similar frame index for the caller-allocated save area
2198 // that will be used to store the incoming registers.
2199 int64_t RegSaveOffset =
2200 -SystemZMC::ELFCallFrameSize + TFL->getRegSpillOffset(MF, SystemZ::R2D) - 16;
2201 unsigned RegSaveIndex = MFI.CreateFixedObject(1, RegSaveOffset, true);
2202 FuncInfo->setRegSaveFrameIndex(RegSaveIndex);
2203
2204 // Store the FPR varargs in the reserved frame slots. (We store the
2205 // GPRs as part of the prologue.)
2206 if (NumFixedFPRs < SystemZ::ELFNumArgFPRs && !useSoftFloat()) {
2208 for (unsigned I = NumFixedFPRs; I < SystemZ::ELFNumArgFPRs; ++I) {
2209 unsigned Offset = TFL->getRegSpillOffset(MF, SystemZ::ELFArgFPRs[I]);
2210 int FI =
2212 SDValue FIN = DAG.getFrameIndex(FI, getPointerTy(DAG.getDataLayout()));
2214 &SystemZ::FP64BitRegClass);
2215 SDValue ArgValue = DAG.getCopyFromReg(Chain, DL, VReg, MVT::f64);
2216 MemOps[I] = DAG.getStore(ArgValue.getValue(1), DL, ArgValue, FIN,
2218 }
2219 // Join the stores, which are independent of one another.
2220 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
2221 ArrayRef(&MemOps[NumFixedFPRs],
2222 SystemZ::ELFNumArgFPRs - NumFixedFPRs));
2223 }
2224 }
2225
2226 if (Subtarget.isTargetXPLINK64()) {
2227 // Create virual register for handling incoming "ADA" special register (R5)
2228 const TargetRegisterClass *RC = &SystemZ::ADDR64BitRegClass;
2229 Register ADAvReg = MRI.createVirtualRegister(RC);
2230 auto *Regs = static_cast<SystemZXPLINK64Registers *>(
2231 Subtarget.getSpecialRegisters());
2232 MRI.addLiveIn(Regs->getADARegister(), ADAvReg);
2233 FuncInfo->setADAVirtualRegister(ADAvReg);
2234 }
2235 return Chain;
2236}
2237
2238static bool canUseSiblingCall(const CCState &ArgCCInfo,
2241 // Punt if there are any indirect or stack arguments, or if the call
2242 // needs the callee-saved argument register R6, or if the call uses
2243 // the callee-saved register arguments SwiftSelf and SwiftError.
2244 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2245 CCValAssign &VA = ArgLocs[I];
2247 return false;
2248 if (!VA.isRegLoc())
2249 return false;
2250 Register Reg = VA.getLocReg();
2251 if (Reg == SystemZ::R6H || Reg == SystemZ::R6L || Reg == SystemZ::R6D)
2252 return false;
2253 if (Outs[I].Flags.isSwiftSelf() || Outs[I].Flags.isSwiftError())
2254 return false;
2255 }
2256 return true;
2257}
2258
2260 unsigned Offset, bool LoadAdr = false) {
2263 Register ADAvReg = MFI->getADAVirtualRegister();
2265
2266 SDValue Reg = DAG.getRegister(ADAvReg, PtrVT);
2267 SDValue Ofs = DAG.getTargetConstant(Offset, DL, PtrVT);
2268
2269 SDValue Result = DAG.getNode(SystemZISD::ADA_ENTRY, DL, PtrVT, Val, Reg, Ofs);
2270 if (!LoadAdr)
2271 Result = DAG.getLoad(
2272 PtrVT, DL, DAG.getEntryNode(), Result, MachinePointerInfo(), Align(8),
2274
2275 return Result;
2276}
2277
2278// ADA access using Global value
2279// Note: for functions, address of descriptor is returned
2281 EVT PtrVT) {
2282 unsigned ADAtype;
2283 bool LoadAddr = false;
2284 const GlobalAlias *GA = dyn_cast<GlobalAlias>(GV);
2285 bool IsFunction =
2286 (isa<Function>(GV)) || (GA && isa<Function>(GA->getAliaseeObject()));
2287 bool IsInternal = (GV->hasInternalLinkage() || GV->hasPrivateLinkage());
2288
2289 if (IsFunction) {
2290 if (IsInternal) {
2292 LoadAddr = true;
2293 } else
2295 } else {
2297 }
2298 SDValue Val = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, ADAtype);
2299
2300 return getADAEntry(DAG, Val, DL, 0, LoadAddr);
2301}
2302
2303static bool getzOSCalleeAndADA(SelectionDAG &DAG, SDValue &Callee, SDValue &ADA,
2304 SDLoc &DL, SDValue &Chain) {
2305 unsigned ADADelta = 0; // ADA offset in desc.
2306 unsigned EPADelta = 8; // EPA offset in desc.
2309
2310 // XPLink calling convention.
2311 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
2312 bool IsInternal = (G->getGlobal()->hasInternalLinkage() ||
2313 G->getGlobal()->hasPrivateLinkage());
2314 if (IsInternal) {
2317 Register ADAvReg = MFI->getADAVirtualRegister();
2318 ADA = DAG.getCopyFromReg(Chain, DL, ADAvReg, PtrVT);
2319 Callee = DAG.getTargetGlobalAddress(G->getGlobal(), DL, PtrVT);
2320 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2321 return true;
2322 } else {
2324 G->getGlobal(), DL, PtrVT, 0, SystemZII::MO_ADA_DIRECT_FUNC_DESC);
2325 ADA = getADAEntry(DAG, GA, DL, ADADelta);
2326 Callee = getADAEntry(DAG, GA, DL, EPADelta);
2327 }
2328 } else if (auto *E = dyn_cast<ExternalSymbolSDNode>(Callee)) {
2330 E->getSymbol(), PtrVT, SystemZII::MO_ADA_DIRECT_FUNC_DESC);
2331 ADA = getADAEntry(DAG, ES, DL, ADADelta);
2332 Callee = getADAEntry(DAG, ES, DL, EPADelta);
2333 } else {
2334 // Function pointer case
2335 ADA = DAG.getNode(ISD::ADD, DL, PtrVT, Callee,
2336 DAG.getConstant(ADADelta, DL, PtrVT));
2337 ADA = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), ADA,
2339 Callee = DAG.getNode(ISD::ADD, DL, PtrVT, Callee,
2340 DAG.getConstant(EPADelta, DL, PtrVT));
2341 Callee = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Callee,
2343 }
2344 return false;
2345}
2346
2347SDValue
2349 SmallVectorImpl<SDValue> &InVals) const {
2350 SelectionDAG &DAG = CLI.DAG;
2351 SDLoc &DL = CLI.DL;
2353 SmallVectorImpl<SDValue> &OutVals = CLI.OutVals;
2355 SDValue Chain = CLI.Chain;
2356 SDValue Callee = CLI.Callee;
2357 bool &IsTailCall = CLI.IsTailCall;
2358 CallingConv::ID CallConv = CLI.CallConv;
2359 bool IsVarArg = CLI.IsVarArg;
2361 EVT PtrVT = getPointerTy(MF.getDataLayout());
2362 LLVMContext &Ctx = *DAG.getContext();
2363 SystemZCallingConventionRegisters *Regs = Subtarget.getSpecialRegisters();
2364
2365 // FIXME: z/OS support to be added in later.
2366 if (Subtarget.isTargetXPLINK64())
2367 IsTailCall = false;
2368
2369 // Integer args <=32 bits should have an extension attribute.
2370 verifyNarrowIntegerArgs_Call(Outs, &MF.getFunction(), Callee);
2371
2372 // Analyze the operands of the call, assigning locations to each operand.
2374 CCState ArgCCInfo(CallConv, IsVarArg, MF, ArgLocs, Ctx);
2375 ArgCCInfo.AnalyzeCallOperands(Outs, CC_SystemZ);
2376
2377 // We don't support GuaranteedTailCallOpt, only automatically-detected
2378 // sibling calls.
2379 if (IsTailCall && !canUseSiblingCall(ArgCCInfo, ArgLocs, Outs))
2380 IsTailCall = false;
2381
2382 // Get a count of how many bytes are to be pushed on the stack.
2383 unsigned NumBytes = ArgCCInfo.getStackSize();
2384
2385 // Mark the start of the call.
2386 if (!IsTailCall)
2387 Chain = DAG.getCALLSEQ_START(Chain, NumBytes, 0, DL);
2388
2389 // Copy argument values to their designated locations.
2391 SmallVector<SDValue, 8> MemOpChains;
2392 SDValue StackPtr;
2393 for (unsigned I = 0, E = ArgLocs.size(); I != E; ++I) {
2394 CCValAssign &VA = ArgLocs[I];
2395 SDValue ArgValue = OutVals[I];
2396
2397 if (VA.getLocInfo() == CCValAssign::Indirect) {
2398 // Store the argument in a stack slot and pass its address.
2399 EVT SlotVT;
2400 MVT PartVT;
2401 unsigned NumParts = 1;
2402 if (analyzeArgSplit(Outs, ArgLocs, I, PartVT, NumParts))
2403 SlotVT = EVT::getIntegerVT(Ctx, PartVT.getSizeInBits() * NumParts);
2404 else
2405 SlotVT = Outs[I].VT;
2406 SDValue SpillSlot = DAG.CreateStackTemporary(SlotVT);
2407 int FI = cast<FrameIndexSDNode>(SpillSlot)->getIndex();
2408
2409 MachinePointerInfo StackPtrInfo =
2411 MemOpChains.push_back(
2412 DAG.getStore(Chain, DL, ArgValue, SpillSlot, StackPtrInfo));
2413 // If the original argument was split (e.g. i128), we need
2414 // to store all parts of it here (and pass just one address).
2415 assert(Outs[I].PartOffset == 0);
2416 for (unsigned PartIdx = 1; PartIdx < NumParts; ++PartIdx) {
2417 ++I;
2418 SDValue PartValue = OutVals[I];
2419 unsigned PartOffset = Outs[I].PartOffset;
2420 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, SpillSlot,
2421 DAG.getIntPtrConstant(PartOffset, DL));
2422 MemOpChains.push_back(
2423 DAG.getStore(Chain, DL, PartValue, Address,
2424 StackPtrInfo.getWithOffset(PartOffset)));
2425 assert(PartOffset && "Offset should be non-zero.");
2426 assert((PartOffset + PartValue.getValueType().getStoreSize() <=
2427 SlotVT.getStoreSize()) && "Not enough space for argument part!");
2428 }
2429 ArgValue = SpillSlot;
2430 } else
2431 ArgValue = convertValVTToLocVT(DAG, DL, VA, ArgValue);
2432
2433 if (VA.isRegLoc()) {
2434 // In XPLINK64, for the 128-bit vararg case, ArgValue is bitcasted to a
2435 // MVT::i128 type. We decompose the 128-bit type to a pair of its high
2436 // and low values.
2437 if (VA.getLocVT() == MVT::i128)
2438 ArgValue = lowerI128ToGR128(DAG, ArgValue);
2439 // Queue up the argument copies and emit them at the end.
2440 RegsToPass.push_back(std::make_pair(VA.getLocReg(), ArgValue));
2441 } else {
2442 assert(VA.isMemLoc() && "Argument not register or memory");
2443
2444 // Work out the address of the stack slot. Unpromoted ints and
2445 // floats are passed as right-justified 8-byte values.
2446 if (!StackPtr.getNode())
2447 StackPtr = DAG.getCopyFromReg(Chain, DL,
2448 Regs->getStackPointerRegister(), PtrVT);
2449 unsigned Offset = Regs->getStackPointerBias() + Regs->getCallFrameSize() +
2450 VA.getLocMemOffset();
2451 if (VA.getLocVT() == MVT::i32 || VA.getLocVT() == MVT::f32)
2452 Offset += 4;
2453 else if (VA.getLocVT() == MVT::f16)
2454 Offset += 6;
2455 SDValue Address = DAG.getNode(ISD::ADD, DL, PtrVT, StackPtr,
2457
2458 // Emit the store.
2459 MemOpChains.push_back(
2460 DAG.getStore(Chain, DL, ArgValue, Address, MachinePointerInfo()));
2461
2462 // Although long doubles or vectors are passed through the stack when
2463 // they are vararg (non-fixed arguments), if a long double or vector
2464 // occupies the third and fourth slot of the argument list GPR3 should
2465 // still shadow the third slot of the argument list.
2466 if (Subtarget.isTargetXPLINK64() && VA.needsCustom()) {
2467 SDValue ShadowArgValue =
2468 DAG.getNode(ISD::EXTRACT_ELEMENT, DL, MVT::i64, ArgValue,
2469 DAG.getIntPtrConstant(1, DL));
2470 RegsToPass.push_back(std::make_pair(SystemZ::R3D, ShadowArgValue));
2471 }
2472 }
2473 }
2474
2475 // Join the stores, which are independent of one another.
2476 if (!MemOpChains.empty())
2477 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOpChains);
2478
2479 // Accept direct calls by converting symbolic call addresses to the
2480 // associated Target* opcodes. Force %r1 to be used for indirect
2481 // tail calls.
2482 SDValue Glue;
2483
2484 if (Subtarget.isTargetXPLINK64()) {
2485 SDValue ADA;
2486 bool IsBRASL = getzOSCalleeAndADA(DAG, Callee, ADA, DL, Chain);
2487 if (!IsBRASL) {
2488 unsigned CalleeReg = static_cast<SystemZXPLINK64Registers *>(Regs)
2489 ->getAddressOfCalleeRegister();
2490 Chain = DAG.getCopyToReg(Chain, DL, CalleeReg, Callee, Glue);
2491 Glue = Chain.getValue(1);
2492 Callee = DAG.getRegister(CalleeReg, Callee.getValueType());
2493 }
2494 RegsToPass.push_back(std::make_pair(
2495 static_cast<SystemZXPLINK64Registers *>(Regs)->getADARegister(), ADA));
2496 } else {
2497 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee)) {
2498 Callee = DAG.getTargetGlobalAddress(G->getGlobal(), DL, PtrVT);
2499 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2500 } else if (auto *E = dyn_cast<ExternalSymbolSDNode>(Callee)) {
2501 Callee = DAG.getTargetExternalSymbol(E->getSymbol(), PtrVT);
2502 Callee = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Callee);
2503 } else if (IsTailCall) {
2504 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R1D, Callee, Glue);
2505 Glue = Chain.getValue(1);
2506 Callee = DAG.getRegister(SystemZ::R1D, Callee.getValueType());
2507 }
2508 }
2509
2510 // Build a sequence of copy-to-reg nodes, chained and glued together.
2511 for (const auto &[Reg, N] : RegsToPass) {
2512 Chain = DAG.getCopyToReg(Chain, DL, Reg, N, Glue);
2513 Glue = Chain.getValue(1);
2514 }
2515
2516 // The first call operand is the chain and the second is the target address.
2518 Ops.push_back(Chain);
2519 Ops.push_back(Callee);
2520
2521 // Add argument registers to the end of the list so that they are
2522 // known live into the call.
2523 for (const auto &[Reg, N] : RegsToPass)
2524 Ops.push_back(DAG.getRegister(Reg, N.getValueType()));
2525
2526 // Add a register mask operand representing the call-preserved registers.
2527 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
2528 const uint32_t *Mask = TRI->getCallPreservedMask(MF, CallConv);
2529 assert(Mask && "Missing call preserved mask for calling convention");
2530 Ops.push_back(DAG.getRegisterMask(Mask));
2531
2532 // Glue the call to the argument copies, if any.
2533 if (Glue.getNode())
2534 Ops.push_back(Glue);
2535
2536 // Emit the call.
2537 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
2538 if (IsTailCall) {
2539 SDValue Ret = DAG.getNode(SystemZISD::SIBCALL, DL, NodeTys, Ops);
2540 DAG.addNoMergeSiteInfo(Ret.getNode(), CLI.NoMerge);
2541 return Ret;
2542 }
2543 Chain = DAG.getNode(SystemZISD::CALL, DL, NodeTys, Ops);
2544 DAG.addNoMergeSiteInfo(Chain.getNode(), CLI.NoMerge);
2545 Glue = Chain.getValue(1);
2546
2547 // Mark the end of the call, which is glued to the call itself.
2548 Chain = DAG.getCALLSEQ_END(Chain, NumBytes, 0, Glue, DL);
2549 Glue = Chain.getValue(1);
2550
2551 // Assign locations to each value returned by this call.
2553 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, Ctx);
2554 RetCCInfo.AnalyzeCallResult(Ins, RetCC_SystemZ);
2555
2556 // Copy all of the result registers out of their specified physreg.
2557 for (CCValAssign &VA : RetLocs) {
2558 // Copy the value out, gluing the copy to the end of the call sequence.
2559 SDValue RetValue = DAG.getCopyFromReg(Chain, DL, VA.getLocReg(),
2560 VA.getLocVT(), Glue);
2561 Chain = RetValue.getValue(1);
2562 Glue = RetValue.getValue(2);
2563
2564 // Convert the value of the return register into the value that's
2565 // being returned.
2566 InVals.push_back(convertLocVTToValVT(DAG, DL, VA, Chain, RetValue));
2567 }
2568
2569 return Chain;
2570}
2571
2572// Generate a call taking the given operands as arguments and returning a
2573// result of type RetVT.
2575 SDValue Chain, SelectionDAG &DAG, const char *CalleeName, EVT RetVT,
2576 ArrayRef<SDValue> Ops, CallingConv::ID CallConv, bool IsSigned, SDLoc DL,
2577 bool DoesNotReturn, bool IsReturnValueUsed) const {
2579 Args.reserve(Ops.size());
2580
2581 for (SDValue Op : Ops) {
2583 Op, Op.getValueType().getTypeForEVT(*DAG.getContext()));
2584 Entry.IsSExt = shouldSignExtendTypeInLibCall(Entry.Ty, IsSigned);
2585 Entry.IsZExt = !Entry.IsSExt;
2586 Args.push_back(Entry);
2587 }
2588
2589 SDValue Callee =
2590 DAG.getExternalSymbol(CalleeName, getPointerTy(DAG.getDataLayout()));
2591
2592 Type *RetTy = RetVT.getTypeForEVT(*DAG.getContext());
2594 bool SignExtend = shouldSignExtendTypeInLibCall(RetTy, IsSigned);
2595 CLI.setDebugLoc(DL)
2596 .setChain(Chain)
2597 .setCallee(CallConv, RetTy, Callee, std::move(Args))
2598 .setNoReturn(DoesNotReturn)
2599 .setDiscardResult(!IsReturnValueUsed)
2600 .setSExtResult(SignExtend)
2601 .setZExtResult(!SignExtend);
2602 return LowerCallTo(CLI);
2603}
2604
2606 CallingConv::ID CallConv, MachineFunction &MF, bool IsVarArg,
2607 const SmallVectorImpl<ISD::OutputArg> &Outs, LLVMContext &Context,
2608 const Type *RetTy) const {
2609 // Special case that we cannot easily detect in RetCC_SystemZ since
2610 // i128 may not be a legal type.
2611 for (auto &Out : Outs)
2612 if (Out.ArgVT.isScalarInteger() && Out.ArgVT.getSizeInBits() > 64)
2613 return false;
2614
2616 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, Context);
2617 return RetCCInfo.CheckReturn(Outs, RetCC_SystemZ);
2618}
2619
2620SDValue
2622 bool IsVarArg,
2624 const SmallVectorImpl<SDValue> &OutVals,
2625 const SDLoc &DL, SelectionDAG &DAG) const {
2627
2628 // Integer args <=32 bits should have an extension attribute.
2629 verifyNarrowIntegerArgs_Ret(Outs, &MF.getFunction());
2630
2631 // Assign locations to each returned value.
2633 CCState RetCCInfo(CallConv, IsVarArg, MF, RetLocs, *DAG.getContext());
2634 RetCCInfo.AnalyzeReturn(Outs, RetCC_SystemZ);
2635
2636 // Quick exit for void returns
2637 if (RetLocs.empty())
2638 return DAG.getNode(SystemZISD::RET_GLUE, DL, MVT::Other, Chain);
2639
2640 if (CallConv == CallingConv::GHC)
2641 report_fatal_error("GHC functions return void only");
2642
2643 // Copy the result values into the output registers.
2644 SDValue Glue;
2646 RetOps.push_back(Chain);
2647 for (unsigned I = 0, E = RetLocs.size(); I != E; ++I) {
2648 CCValAssign &VA = RetLocs[I];
2649 SDValue RetValue = OutVals[I];
2650
2651 // Make the return register live on exit.
2652 assert(VA.isRegLoc() && "Can only return in registers!");
2653
2654 // Promote the value as required.
2655 RetValue = convertValVTToLocVT(DAG, DL, VA, RetValue);
2656
2657 // Chain and glue the copies together.
2658 Register Reg = VA.getLocReg();
2659 Chain = DAG.getCopyToReg(Chain, DL, Reg, RetValue, Glue);
2660 Glue = Chain.getValue(1);
2661 RetOps.push_back(DAG.getRegister(Reg, VA.getLocVT()));
2662 }
2663
2664 // Update chain and glue.
2665 RetOps[0] = Chain;
2666 if (Glue.getNode())
2667 RetOps.push_back(Glue);
2668
2669 return DAG.getNode(SystemZISD::RET_GLUE, DL, MVT::Other, RetOps);
2670}
2671
2672// Return true if Op is an intrinsic node with chain that returns the CC value
2673// as its only (other) argument. Provide the associated SystemZISD opcode and
2674// the mask of valid CC values if so.
2675static bool isIntrinsicWithCCAndChain(SDValue Op, unsigned &Opcode,
2676 unsigned &CCValid) {
2677 unsigned Id = Op.getConstantOperandVal(1);
2678 switch (Id) {
2679 case Intrinsic::s390_tbegin:
2680 Opcode = SystemZISD::TBEGIN;
2681 CCValid = SystemZ::CCMASK_TBEGIN;
2682 return true;
2683
2684 case Intrinsic::s390_tbegin_nofloat:
2685 Opcode = SystemZISD::TBEGIN_NOFLOAT;
2686 CCValid = SystemZ::CCMASK_TBEGIN;
2687 return true;
2688
2689 case Intrinsic::s390_tend:
2690 Opcode = SystemZISD::TEND;
2691 CCValid = SystemZ::CCMASK_TEND;
2692 return true;
2693
2694 default:
2695 return false;
2696 }
2697}
2698
2699// Return true if Op is an intrinsic node without chain that returns the
2700// CC value as its final argument. Provide the associated SystemZISD
2701// opcode and the mask of valid CC values if so.
2702static bool isIntrinsicWithCC(SDValue Op, unsigned &Opcode, unsigned &CCValid) {
2703 unsigned Id = Op.getConstantOperandVal(0);
2704 switch (Id) {
2705 case Intrinsic::s390_vpkshs:
2706 case Intrinsic::s390_vpksfs:
2707 case Intrinsic::s390_vpksgs:
2708 Opcode = SystemZISD::PACKS_CC;
2709 CCValid = SystemZ::CCMASK_VCMP;
2710 return true;
2711
2712 case Intrinsic::s390_vpklshs:
2713 case Intrinsic::s390_vpklsfs:
2714 case Intrinsic::s390_vpklsgs:
2715 Opcode = SystemZISD::PACKLS_CC;
2716 CCValid = SystemZ::CCMASK_VCMP;
2717 return true;
2718
2719 case Intrinsic::s390_vceqbs:
2720 case Intrinsic::s390_vceqhs:
2721 case Intrinsic::s390_vceqfs:
2722 case Intrinsic::s390_vceqgs:
2723 case Intrinsic::s390_vceqqs:
2724 Opcode = SystemZISD::VICMPES;
2725 CCValid = SystemZ::CCMASK_VCMP;
2726 return true;
2727
2728 case Intrinsic::s390_vchbs:
2729 case Intrinsic::s390_vchhs:
2730 case Intrinsic::s390_vchfs:
2731 case Intrinsic::s390_vchgs:
2732 case Intrinsic::s390_vchqs:
2733 Opcode = SystemZISD::VICMPHS;
2734 CCValid = SystemZ::CCMASK_VCMP;
2735 return true;
2736
2737 case Intrinsic::s390_vchlbs:
2738 case Intrinsic::s390_vchlhs:
2739 case Intrinsic::s390_vchlfs:
2740 case Intrinsic::s390_vchlgs:
2741 case Intrinsic::s390_vchlqs:
2742 Opcode = SystemZISD::VICMPHLS;
2743 CCValid = SystemZ::CCMASK_VCMP;
2744 return true;
2745
2746 case Intrinsic::s390_vtm:
2747 Opcode = SystemZISD::VTM;
2748 CCValid = SystemZ::CCMASK_VCMP;
2749 return true;
2750
2751 case Intrinsic::s390_vfaebs:
2752 case Intrinsic::s390_vfaehs:
2753 case Intrinsic::s390_vfaefs:
2754 Opcode = SystemZISD::VFAE_CC;
2755 CCValid = SystemZ::CCMASK_ANY;
2756 return true;
2757
2758 case Intrinsic::s390_vfaezbs:
2759 case Intrinsic::s390_vfaezhs:
2760 case Intrinsic::s390_vfaezfs:
2761 Opcode = SystemZISD::VFAEZ_CC;
2762 CCValid = SystemZ::CCMASK_ANY;
2763 return true;
2764
2765 case Intrinsic::s390_vfeebs:
2766 case Intrinsic::s390_vfeehs:
2767 case Intrinsic::s390_vfeefs:
2768 Opcode = SystemZISD::VFEE_CC;
2769 CCValid = SystemZ::CCMASK_ANY;
2770 return true;
2771
2772 case Intrinsic::s390_vfeezbs:
2773 case Intrinsic::s390_vfeezhs:
2774 case Intrinsic::s390_vfeezfs:
2775 Opcode = SystemZISD::VFEEZ_CC;
2776 CCValid = SystemZ::CCMASK_ANY;
2777 return true;
2778
2779 case Intrinsic::s390_vfenebs:
2780 case Intrinsic::s390_vfenehs:
2781 case Intrinsic::s390_vfenefs:
2782 Opcode = SystemZISD::VFENE_CC;
2783 CCValid = SystemZ::CCMASK_ANY;
2784 return true;
2785
2786 case Intrinsic::s390_vfenezbs:
2787 case Intrinsic::s390_vfenezhs:
2788 case Intrinsic::s390_vfenezfs:
2789 Opcode = SystemZISD::VFENEZ_CC;
2790 CCValid = SystemZ::CCMASK_ANY;
2791 return true;
2792
2793 case Intrinsic::s390_vistrbs:
2794 case Intrinsic::s390_vistrhs:
2795 case Intrinsic::s390_vistrfs:
2796 Opcode = SystemZISD::VISTR_CC;
2798 return true;
2799
2800 case Intrinsic::s390_vstrcbs:
2801 case Intrinsic::s390_vstrchs:
2802 case Intrinsic::s390_vstrcfs:
2803 Opcode = SystemZISD::VSTRC_CC;
2804 CCValid = SystemZ::CCMASK_ANY;
2805 return true;
2806
2807 case Intrinsic::s390_vstrczbs:
2808 case Intrinsic::s390_vstrczhs:
2809 case Intrinsic::s390_vstrczfs:
2810 Opcode = SystemZISD::VSTRCZ_CC;
2811 CCValid = SystemZ::CCMASK_ANY;
2812 return true;
2813
2814 case Intrinsic::s390_vstrsb:
2815 case Intrinsic::s390_vstrsh:
2816 case Intrinsic::s390_vstrsf:
2817 Opcode = SystemZISD::VSTRS_CC;
2818 CCValid = SystemZ::CCMASK_ANY;
2819 return true;
2820
2821 case Intrinsic::s390_vstrszb:
2822 case Intrinsic::s390_vstrszh:
2823 case Intrinsic::s390_vstrszf:
2824 Opcode = SystemZISD::VSTRSZ_CC;
2825 CCValid = SystemZ::CCMASK_ANY;
2826 return true;
2827
2828 case Intrinsic::s390_vfcedbs:
2829 case Intrinsic::s390_vfcesbs:
2830 Opcode = SystemZISD::VFCMPES;
2831 CCValid = SystemZ::CCMASK_VCMP;
2832 return true;
2833
2834 case Intrinsic::s390_vfchdbs:
2835 case Intrinsic::s390_vfchsbs:
2836 Opcode = SystemZISD::VFCMPHS;
2837 CCValid = SystemZ::CCMASK_VCMP;
2838 return true;
2839
2840 case Intrinsic::s390_vfchedbs:
2841 case Intrinsic::s390_vfchesbs:
2842 Opcode = SystemZISD::VFCMPHES;
2843 CCValid = SystemZ::CCMASK_VCMP;
2844 return true;
2845
2846 case Intrinsic::s390_vftcidb:
2847 case Intrinsic::s390_vftcisb:
2848 Opcode = SystemZISD::VFTCI;
2849 CCValid = SystemZ::CCMASK_VCMP;
2850 return true;
2851
2852 case Intrinsic::s390_tdc:
2853 Opcode = SystemZISD::TDC;
2854 CCValid = SystemZ::CCMASK_TDC;
2855 return true;
2856
2857 default:
2858 return false;
2859 }
2860}
2861
2862// Emit an intrinsic with chain and an explicit CC register result.
2864 unsigned Opcode) {
2865 // Copy all operands except the intrinsic ID.
2866 unsigned NumOps = Op.getNumOperands();
2868 Ops.reserve(NumOps - 1);
2869 Ops.push_back(Op.getOperand(0));
2870 for (unsigned I = 2; I < NumOps; ++I)
2871 Ops.push_back(Op.getOperand(I));
2872
2873 assert(Op->getNumValues() == 2 && "Expected only CC result and chain");
2874 SDVTList RawVTs = DAG.getVTList(MVT::i32, MVT::Other);
2875 SDValue Intr = DAG.getNode(Opcode, SDLoc(Op), RawVTs, Ops);
2876 SDValue OldChain = SDValue(Op.getNode(), 1);
2877 SDValue NewChain = SDValue(Intr.getNode(), 1);
2878 DAG.ReplaceAllUsesOfValueWith(OldChain, NewChain);
2879 return Intr.getNode();
2880}
2881
2882// Emit an intrinsic with an explicit CC register result.
2884 unsigned Opcode) {
2885 // Copy all operands except the intrinsic ID.
2886 SDLoc DL(Op);
2887 unsigned NumOps = Op.getNumOperands();
2889 Ops.reserve(NumOps - 1);
2890 for (unsigned I = 1; I < NumOps; ++I) {
2891 SDValue CurrOper = Op.getOperand(I);
2892 if (CurrOper.getValueType() == MVT::f16) {
2893 assert((Op.getConstantOperandVal(0) == Intrinsic::s390_tdc && I == 1) &&
2894 "Unhandled intrinsic with f16 operand.");
2895 CurrOper = DAG.getFPExtendOrRound(CurrOper, DL, MVT::f32);
2896 }
2897 Ops.push_back(CurrOper);
2898 }
2899
2900 SDValue Intr = DAG.getNode(Opcode, DL, Op->getVTList(), Ops);
2901 return Intr.getNode();
2902}
2903
2904// CC is a comparison that will be implemented using an integer or
2905// floating-point comparison. Return the condition code mask for
2906// a branch on true. In the integer case, CCMASK_CMP_UO is set for
2907// unsigned comparisons and clear for signed ones. In the floating-point
2908// case, CCMASK_CMP_UO has its normal mask meaning (unordered).
2910#define CONV(X) \
2911 case ISD::SET##X: return SystemZ::CCMASK_CMP_##X; \
2912 case ISD::SETO##X: return SystemZ::CCMASK_CMP_##X; \
2913 case ISD::SETU##X: return SystemZ::CCMASK_CMP_UO | SystemZ::CCMASK_CMP_##X
2914
2915 switch (CC) {
2916 default:
2917 llvm_unreachable("Invalid integer condition!");
2918
2919 CONV(EQ);
2920 CONV(NE);
2921 CONV(GT);
2922 CONV(GE);
2923 CONV(LT);
2924 CONV(LE);
2925
2926 case ISD::SETO: return SystemZ::CCMASK_CMP_O;
2928 }
2929#undef CONV
2930}
2931
2932// If C can be converted to a comparison against zero, adjust the operands
2933// as necessary.
2934static void adjustZeroCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C) {
2935 if (C.ICmpType == SystemZICMP::UnsignedOnly)
2936 return;
2937
2938 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1.getNode());
2939 if (!ConstOp1 || ConstOp1->getValueSizeInBits(0) > 64)
2940 return;
2941
2942 int64_t Value = ConstOp1->getSExtValue();
2943 if ((Value == -1 && C.CCMask == SystemZ::CCMASK_CMP_GT) ||
2944 (Value == -1 && C.CCMask == SystemZ::CCMASK_CMP_LE) ||
2945 (Value == 1 && C.CCMask == SystemZ::CCMASK_CMP_LT) ||
2946 (Value == 1 && C.CCMask == SystemZ::CCMASK_CMP_GE)) {
2947 C.CCMask ^= SystemZ::CCMASK_CMP_EQ;
2948 C.Op1 = DAG.getConstant(0, DL, C.Op1.getValueType());
2949 }
2950}
2951
2952// If a comparison described by C is suitable for CLI(Y), CHHSI or CLHHSI,
2953// adjust the operands as necessary.
2954static void adjustSubwordCmp(SelectionDAG &DAG, const SDLoc &DL,
2955 Comparison &C) {
2956 // For us to make any changes, it must a comparison between a single-use
2957 // load and a constant.
2958 if (!C.Op0.hasOneUse() ||
2959 C.Op0.getOpcode() != ISD::LOAD ||
2960 C.Op1.getOpcode() != ISD::Constant)
2961 return;
2962
2963 // We must have an 8- or 16-bit load.
2964 auto *Load = cast<LoadSDNode>(C.Op0);
2965 unsigned NumBits = Load->getMemoryVT().getSizeInBits();
2966 if ((NumBits != 8 && NumBits != 16) ||
2967 NumBits != Load->getMemoryVT().getStoreSizeInBits())
2968 return;
2969
2970 // The load must be an extending one and the constant must be within the
2971 // range of the unextended value.
2972 auto *ConstOp1 = cast<ConstantSDNode>(C.Op1);
2973 if (!ConstOp1 || ConstOp1->getValueSizeInBits(0) > 64)
2974 return;
2975 uint64_t Value = ConstOp1->getZExtValue();
2976 uint64_t Mask = (1 << NumBits) - 1;
2977 if (Load->getExtensionType() == ISD::SEXTLOAD) {
2978 // Make sure that ConstOp1 is in range of C.Op0.
2979 int64_t SignedValue = ConstOp1->getSExtValue();
2980 if (uint64_t(SignedValue) + (uint64_t(1) << (NumBits - 1)) > Mask)
2981 return;
2982 if (C.ICmpType != SystemZICMP::SignedOnly) {
2983 // Unsigned comparison between two sign-extended values is equivalent
2984 // to unsigned comparison between two zero-extended values.
2985 Value &= Mask;
2986 } else if (NumBits == 8) {
2987 // Try to treat the comparison as unsigned, so that we can use CLI.
2988 // Adjust CCMask and Value as necessary.
2989 if (Value == 0 && C.CCMask == SystemZ::CCMASK_CMP_LT)
2990 // Test whether the high bit of the byte is set.
2991 Value = 127, C.CCMask = SystemZ::CCMASK_CMP_GT;
2992 else if (Value == 0 && C.CCMask == SystemZ::CCMASK_CMP_GE)
2993 // Test whether the high bit of the byte is clear.
2994 Value = 128, C.CCMask = SystemZ::CCMASK_CMP_LT;
2995 else
2996 // No instruction exists for this combination.
2997 return;
2998 C.ICmpType = SystemZICMP::UnsignedOnly;
2999 }
3000 } else if (Load->getExtensionType() == ISD::ZEXTLOAD) {
3001 if (Value > Mask)
3002 return;
3003 // If the constant is in range, we can use any comparison.
3004 C.ICmpType = SystemZICMP::Any;
3005 } else
3006 return;
3007
3008 // Make sure that the first operand is an i32 of the right extension type.
3009 ISD::LoadExtType ExtType = (C.ICmpType == SystemZICMP::SignedOnly ?
3012 if (C.Op0.getValueType() != MVT::i32 ||
3013 Load->getExtensionType() != ExtType) {
3014 C.Op0 = DAG.getExtLoad(ExtType, SDLoc(Load), MVT::i32, Load->getChain(),
3015 Load->getBasePtr(), Load->getPointerInfo(),
3016 Load->getMemoryVT(), Load->getAlign(),
3017 Load->getMemOperand()->getFlags());
3018 // Update the chain uses.
3019 DAG.ReplaceAllUsesOfValueWith(SDValue(Load, 1), C.Op0.getValue(1));
3020 }
3021
3022 // Make sure that the second operand is an i32 with the right value.
3023 if (C.Op1.getValueType() != MVT::i32 ||
3024 Value != ConstOp1->getZExtValue())
3025 C.Op1 = DAG.getConstant((uint32_t)Value, DL, MVT::i32);
3026}
3027
3028// Return true if Op is either an unextended load, or a load suitable
3029// for integer register-memory comparisons of type ICmpType.
3030static bool isNaturalMemoryOperand(SDValue Op, unsigned ICmpType) {
3031 auto *Load = dyn_cast<LoadSDNode>(Op.getNode());
3032 if (Load) {
3033 // There are no instructions to compare a register with a memory byte.
3034 if (Load->getMemoryVT() == MVT::i8)
3035 return false;
3036 // Otherwise decide on extension type.
3037 switch (Load->getExtensionType()) {
3038 case ISD::NON_EXTLOAD:
3039 return true;
3040 case ISD::SEXTLOAD:
3041 return ICmpType != SystemZICMP::UnsignedOnly;
3042 case ISD::ZEXTLOAD:
3043 return ICmpType != SystemZICMP::SignedOnly;
3044 default:
3045 break;
3046 }
3047 }
3048 return false;
3049}
3050
3051// Return true if it is better to swap the operands of C.
3052static bool shouldSwapCmpOperands(const Comparison &C) {
3053 // If one side of the compare is a load of the stackguard reference value,
3054 // then that load should be Op1.
3055 if (C.Op0.isMachineOpcode() &&
3056 (C.Op0.getMachineOpcode() == SystemZ::LOAD_STACK_GUARD))
3057 return true;
3058
3059 // Leave i128 and f128 comparisons alone, since they have no memory forms.
3060 if (C.Op0.getValueType() == MVT::i128)
3061 return false;
3062 if (C.Op0.getValueType() == MVT::f128)
3063 return false;
3064
3065 // Always keep a floating-point constant second, since comparisons with
3066 // zero can use LOAD TEST and comparisons with other constants make a
3067 // natural memory operand.
3068 if (isa<ConstantFPSDNode>(C.Op1))
3069 return false;
3070
3071 // Never swap comparisons with zero since there are many ways to optimize
3072 // those later.
3073 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1);
3074 if (ConstOp1 && ConstOp1->getZExtValue() == 0)
3075 return false;
3076
3077 // Also keep natural memory operands second if the loaded value is
3078 // only used here. Several comparisons have memory forms.
3079 if (isNaturalMemoryOperand(C.Op1, C.ICmpType) && C.Op1.hasOneUse())
3080 return false;
3081
3082 // Look for cases where Cmp0 is a single-use load and Cmp1 isn't.
3083 // In that case we generally prefer the memory to be second.
3084 if (isNaturalMemoryOperand(C.Op0, C.ICmpType) && C.Op0.hasOneUse()) {
3085 // The only exceptions are when the second operand is a constant and
3086 // we can use things like CHHSI.
3087 if (!ConstOp1)
3088 return true;
3089 // The unsigned memory-immediate instructions can handle 16-bit
3090 // unsigned integers.
3091 if (C.ICmpType != SystemZICMP::SignedOnly &&
3092 isUInt<16>(ConstOp1->getZExtValue()))
3093 return false;
3094 // The signed memory-immediate instructions can handle 16-bit
3095 // signed integers.
3096 if (C.ICmpType != SystemZICMP::UnsignedOnly &&
3097 isInt<16>(ConstOp1->getSExtValue()))
3098 return false;
3099 return true;
3100 }
3101
3102 // Try to promote the use of CGFR and CLGFR.
3103 unsigned Opcode0 = C.Op0.getOpcode();
3104 if (C.ICmpType != SystemZICMP::UnsignedOnly && Opcode0 == ISD::SIGN_EXTEND)
3105 return true;
3106 if (C.ICmpType != SystemZICMP::SignedOnly && Opcode0 == ISD::ZERO_EXTEND)
3107 return true;
3108 if (C.ICmpType != SystemZICMP::SignedOnly && Opcode0 == ISD::AND &&
3109 C.Op0.getOperand(1).getOpcode() == ISD::Constant &&
3110 C.Op0.getConstantOperandVal(1) == 0xffffffff)
3111 return true;
3112
3113 return false;
3114}
3115
3116// Check whether C tests for equality between X and Y and whether X - Y
3117// or Y - X is also computed. In that case it's better to compare the
3118// result of the subtraction against zero.
3120 Comparison &C) {
3121 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3122 C.CCMask == SystemZ::CCMASK_CMP_NE) {
3123 for (SDNode *N : C.Op0->users()) {
3124 if (N->getOpcode() == ISD::SUB &&
3125 ((N->getOperand(0) == C.Op0 && N->getOperand(1) == C.Op1) ||
3126 (N->getOperand(0) == C.Op1 && N->getOperand(1) == C.Op0))) {
3127 // Disable the nsw and nuw flags: the backend needs to handle
3128 // overflow as well during comparison elimination.
3129 N->dropFlags(SDNodeFlags::NoWrap);
3130 C.Op0 = SDValue(N, 0);
3131 C.Op1 = DAG.getConstant(0, DL, N->getValueType(0));
3132 return;
3133 }
3134 }
3135 }
3136}
3137
3138// Check whether C compares a floating-point value with zero and if that
3139// floating-point value is also negated. In this case we can use the
3140// negation to set CC, so avoiding separate LOAD AND TEST and
3141// LOAD (NEGATIVE/COMPLEMENT) instructions.
3142static void adjustForFNeg(Comparison &C) {
3143 // This optimization is invalid for strict comparisons, since FNEG
3144 // does not raise any exceptions.
3145 if (C.Chain)
3146 return;
3147 auto *C1 = dyn_cast<ConstantFPSDNode>(C.Op1);
3148 if (C1 && C1->isZero()) {
3149 for (SDNode *N : C.Op0->users()) {
3150 if (N->getOpcode() == ISD::FNEG) {
3151 C.Op0 = SDValue(N, 0);
3152 C.CCMask = SystemZ::reverseCCMask(C.CCMask);
3153 return;
3154 }
3155 }
3156 }
3157}
3158
3159// Check whether C compares (shl X, 32) with 0 and whether X is
3160// also sign-extended. In that case it is better to test the result
3161// of the sign extension using LTGFR.
3162//
3163// This case is important because InstCombine transforms a comparison
3164// with (sext (trunc X)) into a comparison with (shl X, 32).
3165static void adjustForLTGFR(Comparison &C) {
3166 // Check for a comparison between (shl X, 32) and 0.
3167 if (C.Op0.getOpcode() == ISD::SHL && C.Op0.getValueType() == MVT::i64 &&
3168 C.Op1.getOpcode() == ISD::Constant && C.Op1->getAsZExtVal() == 0) {
3169 auto *C1 = dyn_cast<ConstantSDNode>(C.Op0.getOperand(1));
3170 if (C1 && C1->getZExtValue() == 32) {
3171 SDValue ShlOp0 = C.Op0.getOperand(0);
3172 // See whether X has any SIGN_EXTEND_INREG uses.
3173 for (SDNode *N : ShlOp0->users()) {
3174 if (N->getOpcode() == ISD::SIGN_EXTEND_INREG &&
3175 cast<VTSDNode>(N->getOperand(1))->getVT() == MVT::i32) {
3176 C.Op0 = SDValue(N, 0);
3177 return;
3178 }
3179 }
3180 }
3181 }
3182}
3183
3184// If C compares the truncation of an extending load, try to compare
3185// the untruncated value instead. This exposes more opportunities to
3186// reuse CC.
3187static void adjustICmpTruncate(SelectionDAG &DAG, const SDLoc &DL,
3188 Comparison &C) {
3189 if (C.Op0.getOpcode() == ISD::TRUNCATE &&
3190 C.Op0.getOperand(0).getOpcode() == ISD::LOAD &&
3191 C.Op1.getOpcode() == ISD::Constant &&
3192 cast<ConstantSDNode>(C.Op1)->getValueSizeInBits(0) <= 64 &&
3193 C.Op1->getAsZExtVal() == 0) {
3194 auto *L = cast<LoadSDNode>(C.Op0.getOperand(0));
3195 if (L->getMemoryVT().getStoreSizeInBits().getFixedValue() <=
3196 C.Op0.getValueSizeInBits().getFixedValue()) {
3197 unsigned Type = L->getExtensionType();
3198 if ((Type == ISD::ZEXTLOAD && C.ICmpType != SystemZICMP::SignedOnly) ||
3199 (Type == ISD::SEXTLOAD && C.ICmpType != SystemZICMP::UnsignedOnly)) {
3200 C.Op0 = C.Op0.getOperand(0);
3201 C.Op1 = DAG.getConstant(0, DL, C.Op0.getValueType());
3202 }
3203 }
3204 }
3205}
3206
3207// Adjust if a given Compare is a check of the stack guard against a stack
3208// guard instance on the stack. Specifically, this checks if:
3209// - The operands are a load of the stack guard, and a load from a stack slot
3210// - The original opcode is ICMP
3211// - ICMPType is compatible with unsigned comparison.
3213 Comparison &C) {
3214
3215 // Opcode must be ICMP.
3216 if (C.Opcode != SystemZISD::ICMP)
3217 return;
3218 // ICmpType must be Unsigned or Any.
3219 if (C.ICmpType == SystemZICMP::SignedOnly)
3220 return;
3221 // Op0 must be FrameIndex Load.
3222 if (!(ISD::isNormalLoad(C.Op0.getNode()) &&
3223 dyn_cast<FrameIndexSDNode>(C.Op0.getOperand(1))))
3224 return;
3225 // Op1 must be LOAD_STACK_GUARD.
3226 if (!C.Op1.isMachineOpcode() ||
3227 C.Op1.getMachineOpcode() != SystemZ::LOAD_STACK_GUARD)
3228 return;
3229
3230 // At this point we are sure that this is a proper CMP_STACKGUARD
3231 // case, update the opcode to reflect this.
3232 C.Opcode = SystemZISD::CMP_STACKGUARD;
3233 C.Op1 = SDValue();
3234}
3235
3236// Return true if shift operation N has an in-range constant shift value.
3237// Store it in ShiftVal if so.
3238static bool isSimpleShift(SDValue N, unsigned &ShiftVal) {
3239 auto *Shift = dyn_cast<ConstantSDNode>(N.getOperand(1));
3240 if (!Shift)
3241 return false;
3242
3243 uint64_t Amount = Shift->getZExtValue();
3244 if (Amount >= N.getValueSizeInBits())
3245 return false;
3246
3247 ShiftVal = Amount;
3248 return true;
3249}
3250
3251// Check whether an AND with Mask is suitable for a TEST UNDER MASK
3252// instruction and whether the CC value is descriptive enough to handle
3253// a comparison of type Opcode between the AND result and CmpVal.
3254// CCMask says which comparison result is being tested and BitSize is
3255// the number of bits in the operands. If TEST UNDER MASK can be used,
3256// return the corresponding CC mask, otherwise return 0.
3257static unsigned getTestUnderMaskCond(unsigned BitSize, unsigned CCMask,
3258 uint64_t Mask, uint64_t CmpVal,
3259 unsigned ICmpType) {
3260 assert(Mask != 0 && "ANDs with zero should have been removed by now");
3261
3262 // Check whether the mask is suitable for TMHH, TMHL, TMLH or TMLL.
3263 if (!SystemZ::isImmLL(Mask) && !SystemZ::isImmLH(Mask) &&
3264 !SystemZ::isImmHL(Mask) && !SystemZ::isImmHH(Mask))
3265 return 0;
3266
3267 // Work out the masks for the lowest and highest bits.
3269 uint64_t Low = uint64_t(1) << llvm::countr_zero(Mask);
3270
3271 // Signed ordered comparisons are effectively unsigned if the sign
3272 // bit is dropped.
3273 bool EffectivelyUnsigned = (ICmpType != SystemZICMP::SignedOnly);
3274
3275 // Check for equality comparisons with 0, or the equivalent.
3276 if (CmpVal == 0) {
3277 if (CCMask == SystemZ::CCMASK_CMP_EQ)
3279 if (CCMask == SystemZ::CCMASK_CMP_NE)
3281 }
3282 if (EffectivelyUnsigned && CmpVal > 0 && CmpVal <= Low) {
3283 if (CCMask == SystemZ::CCMASK_CMP_LT)
3285 if (CCMask == SystemZ::CCMASK_CMP_GE)
3287 }
3288 if (EffectivelyUnsigned && CmpVal < Low) {
3289 if (CCMask == SystemZ::CCMASK_CMP_LE)
3291 if (CCMask == SystemZ::CCMASK_CMP_GT)
3293 }
3294
3295 // Check for equality comparisons with the mask, or the equivalent.
3296 if (CmpVal == Mask) {
3297 if (CCMask == SystemZ::CCMASK_CMP_EQ)
3299 if (CCMask == SystemZ::CCMASK_CMP_NE)
3301 }
3302 if (EffectivelyUnsigned && CmpVal >= Mask - Low && CmpVal < Mask) {
3303 if (CCMask == SystemZ::CCMASK_CMP_GT)
3305 if (CCMask == SystemZ::CCMASK_CMP_LE)
3307 }
3308 if (EffectivelyUnsigned && CmpVal > Mask - Low && CmpVal <= Mask) {
3309 if (CCMask == SystemZ::CCMASK_CMP_GE)
3311 if (CCMask == SystemZ::CCMASK_CMP_LT)
3313 }
3314
3315 // Check for ordered comparisons with the top bit.
3316 if (EffectivelyUnsigned && CmpVal >= Mask - High && CmpVal < High) {
3317 if (CCMask == SystemZ::CCMASK_CMP_LE)
3319 if (CCMask == SystemZ::CCMASK_CMP_GT)
3321 }
3322 if (EffectivelyUnsigned && CmpVal > Mask - High && CmpVal <= High) {
3323 if (CCMask == SystemZ::CCMASK_CMP_LT)
3325 if (CCMask == SystemZ::CCMASK_CMP_GE)
3327 }
3328
3329 // If there are just two bits, we can do equality checks for Low and High
3330 // as well.
3331 if (Mask == Low + High) {
3332 if (CCMask == SystemZ::CCMASK_CMP_EQ && CmpVal == Low)
3334 if (CCMask == SystemZ::CCMASK_CMP_NE && CmpVal == Low)
3336 if (CCMask == SystemZ::CCMASK_CMP_EQ && CmpVal == High)
3338 if (CCMask == SystemZ::CCMASK_CMP_NE && CmpVal == High)
3340 }
3341
3342 // Looks like we've exhausted our options.
3343 return 0;
3344}
3345
3346// See whether C can be implemented as a TEST UNDER MASK instruction.
3347// Update the arguments with the TM version if so.
3349 Comparison &C) {
3350 // Use VECTOR TEST UNDER MASK for i128 operations.
3351 if (C.Op0.getValueType() == MVT::i128) {
3352 // We can use VTM for EQ/NE comparisons of x & y against 0.
3353 if (C.Op0.getOpcode() == ISD::AND &&
3354 (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3355 C.CCMask == SystemZ::CCMASK_CMP_NE)) {
3356 auto *Mask = dyn_cast<ConstantSDNode>(C.Op1);
3357 if (Mask && Mask->getAPIntValue() == 0) {
3358 C.Opcode = SystemZISD::VTM;
3359 C.Op1 = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, C.Op0.getOperand(1));
3360 C.Op0 = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, C.Op0.getOperand(0));
3361 C.CCValid = SystemZ::CCMASK_VCMP;
3362 if (C.CCMask == SystemZ::CCMASK_CMP_EQ)
3363 C.CCMask = SystemZ::CCMASK_VCMP_ALL;
3364 else
3365 C.CCMask = SystemZ::CCMASK_VCMP_ALL ^ C.CCValid;
3366 }
3367 }
3368 return;
3369 }
3370
3371 // Check that we have a comparison with a constant.
3372 auto *ConstOp1 = dyn_cast<ConstantSDNode>(C.Op1);
3373 if (!ConstOp1)
3374 return;
3375 uint64_t CmpVal = ConstOp1->getZExtValue();
3376
3377 // Check whether the nonconstant input is an AND with a constant mask.
3378 Comparison NewC(C);
3379 uint64_t MaskVal;
3380 ConstantSDNode *Mask = nullptr;
3381 if (C.Op0.getOpcode() == ISD::AND) {
3382 NewC.Op0 = C.Op0.getOperand(0);
3383 NewC.Op1 = C.Op0.getOperand(1);
3384 Mask = dyn_cast<ConstantSDNode>(NewC.Op1);
3385 if (!Mask)
3386 return;
3387 MaskVal = Mask->getZExtValue();
3388 } else {
3389 // There is no instruction to compare with a 64-bit immediate
3390 // so use TMHH instead if possible. We need an unsigned ordered
3391 // comparison with an i64 immediate.
3392 if (NewC.Op0.getValueType() != MVT::i64 ||
3393 NewC.CCMask == SystemZ::CCMASK_CMP_EQ ||
3394 NewC.CCMask == SystemZ::CCMASK_CMP_NE ||
3395 NewC.ICmpType == SystemZICMP::SignedOnly)
3396 return;
3397 // Convert LE and GT comparisons into LT and GE.
3398 if (NewC.CCMask == SystemZ::CCMASK_CMP_LE ||
3399 NewC.CCMask == SystemZ::CCMASK_CMP_GT) {
3400 if (CmpVal == uint64_t(-1))
3401 return;
3402 CmpVal += 1;
3403 NewC.CCMask ^= SystemZ::CCMASK_CMP_EQ;
3404 }
3405 // If the low N bits of Op1 are zero than the low N bits of Op0 can
3406 // be masked off without changing the result.
3407 MaskVal = -(CmpVal & -CmpVal);
3408 NewC.ICmpType = SystemZICMP::UnsignedOnly;
3409 }
3410 if (!MaskVal)
3411 return;
3412
3413 // Check whether the combination of mask, comparison value and comparison
3414 // type are suitable.
3415 unsigned BitSize = NewC.Op0.getValueSizeInBits();
3416 unsigned NewCCMask, ShiftVal;
3417 if (NewC.ICmpType != SystemZICMP::SignedOnly &&
3418 NewC.Op0.getOpcode() == ISD::SHL &&
3419 isSimpleShift(NewC.Op0, ShiftVal) &&
3420 (MaskVal >> ShiftVal != 0) &&
3421 ((CmpVal >> ShiftVal) << ShiftVal) == CmpVal &&
3422 (NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask,
3423 MaskVal >> ShiftVal,
3424 CmpVal >> ShiftVal,
3425 SystemZICMP::Any))) {
3426 NewC.Op0 = NewC.Op0.getOperand(0);
3427 MaskVal >>= ShiftVal;
3428 } else if (NewC.ICmpType != SystemZICMP::SignedOnly &&
3429 NewC.Op0.getOpcode() == ISD::SRL &&
3430 isSimpleShift(NewC.Op0, ShiftVal) &&
3431 (MaskVal << ShiftVal != 0) &&
3432 ((CmpVal << ShiftVal) >> ShiftVal) == CmpVal &&
3433 (NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask,
3434 MaskVal << ShiftVal,
3435 CmpVal << ShiftVal,
3437 NewC.Op0 = NewC.Op0.getOperand(0);
3438 MaskVal <<= ShiftVal;
3439 } else {
3440 NewCCMask = getTestUnderMaskCond(BitSize, NewC.CCMask, MaskVal, CmpVal,
3441 NewC.ICmpType);
3442 if (!NewCCMask)
3443 return;
3444 }
3445
3446 // Go ahead and make the change.
3447 C.Opcode = SystemZISD::TM;
3448 C.Op0 = NewC.Op0;
3449 if (Mask && Mask->getZExtValue() == MaskVal)
3450 C.Op1 = SDValue(Mask, 0);
3451 else
3452 C.Op1 = DAG.getConstant(MaskVal, DL, C.Op0.getValueType());
3453 C.CCValid = SystemZ::CCMASK_TM;
3454 C.CCMask = NewCCMask;
3455}
3456
3457// Implement i128 comparison in vector registers.
3458static void adjustICmp128(SelectionDAG &DAG, const SDLoc &DL,
3459 Comparison &C) {
3460 if (C.Opcode != SystemZISD::ICMP)
3461 return;
3462 if (C.Op0.getValueType() != MVT::i128)
3463 return;
3464
3465 // Recognize vector comparison reductions.
3466 if ((C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3467 C.CCMask == SystemZ::CCMASK_CMP_NE) &&
3468 (isNullConstant(C.Op1) || isAllOnesConstant(C.Op1))) {
3469 bool CmpEq = C.CCMask == SystemZ::CCMASK_CMP_EQ;
3470 bool CmpNull = isNullConstant(C.Op1);
3471 SDValue Src = peekThroughBitcasts(C.Op0);
3472 if (Src.hasOneUse() && isBitwiseNot(Src)) {
3473 Src = Src.getOperand(0);
3474 CmpNull = !CmpNull;
3475 }
3476 unsigned Opcode = 0;
3477 if (Src.hasOneUse()) {
3478 switch (Src.getOpcode()) {
3479 case SystemZISD::VICMPE: Opcode = SystemZISD::VICMPES; break;
3480 case SystemZISD::VICMPH: Opcode = SystemZISD::VICMPHS; break;
3481 case SystemZISD::VICMPHL: Opcode = SystemZISD::VICMPHLS; break;
3482 case SystemZISD::VFCMPE: Opcode = SystemZISD::VFCMPES; break;
3483 case SystemZISD::VFCMPH: Opcode = SystemZISD::VFCMPHS; break;
3484 case SystemZISD::VFCMPHE: Opcode = SystemZISD::VFCMPHES; break;
3485 default: break;
3486 }
3487 }
3488 if (Opcode) {
3489 C.Opcode = Opcode;
3490 C.Op0 = Src->getOperand(0);
3491 C.Op1 = Src->getOperand(1);
3492 C.CCValid = SystemZ::CCMASK_VCMP;
3494 if (!CmpEq)
3495 C.CCMask ^= C.CCValid;
3496 return;
3497 }
3498 }
3499
3500 // Everything below here is not useful if we have native i128 compares.
3501 if (DAG.getSubtarget<SystemZSubtarget>().hasVectorEnhancements3())
3502 return;
3503
3504 // (In-)Equality comparisons can be implemented via VCEQGS.
3505 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3506 C.CCMask == SystemZ::CCMASK_CMP_NE) {
3507 C.Opcode = SystemZISD::VICMPES;
3508 C.Op0 = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, C.Op0);
3509 C.Op1 = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, C.Op1);
3510 C.CCValid = SystemZ::CCMASK_VCMP;
3511 if (C.CCMask == SystemZ::CCMASK_CMP_EQ)
3512 C.CCMask = SystemZ::CCMASK_VCMP_ALL;
3513 else
3514 C.CCMask = SystemZ::CCMASK_VCMP_ALL ^ C.CCValid;
3515 return;
3516 }
3517
3518 // Normalize other comparisons to GT.
3519 bool Swap = false, Invert = false;
3520 switch (C.CCMask) {
3521 case SystemZ::CCMASK_CMP_GT: break;
3522 case SystemZ::CCMASK_CMP_LT: Swap = true; break;
3523 case SystemZ::CCMASK_CMP_LE: Invert = true; break;
3524 case SystemZ::CCMASK_CMP_GE: Swap = Invert = true; break;
3525 default: llvm_unreachable("Invalid integer condition!");
3526 }
3527 if (Swap)
3528 std::swap(C.Op0, C.Op1);
3529
3530 if (C.ICmpType == SystemZICMP::UnsignedOnly)
3531 C.Opcode = SystemZISD::UCMP128HI;
3532 else
3533 C.Opcode = SystemZISD::SCMP128HI;
3534 C.CCValid = SystemZ::CCMASK_ANY;
3535 C.CCMask = SystemZ::CCMASK_1;
3536
3537 if (Invert)
3538 C.CCMask ^= C.CCValid;
3539}
3540
3541// See whether the comparison argument contains a redundant AND
3542// and remove it if so. This sometimes happens due to the generic
3543// BRCOND expansion.
3545 Comparison &C) {
3546 if (C.Op0.getOpcode() != ISD::AND)
3547 return;
3548 auto *Mask = dyn_cast<ConstantSDNode>(C.Op0.getOperand(1));
3549 if (!Mask || Mask->getValueSizeInBits(0) > 64)
3550 return;
3551 KnownBits Known = DAG.computeKnownBits(C.Op0.getOperand(0));
3552 if ((~Known.Zero).getZExtValue() & ~Mask->getZExtValue())
3553 return;
3554
3555 C.Op0 = C.Op0.getOperand(0);
3556}
3557
3558// Return a Comparison that tests the condition-code result of intrinsic
3559// node Call against constant integer CC using comparison code Cond.
3560// Opcode is the opcode of the SystemZISD operation for the intrinsic
3561// and CCValid is the set of possible condition-code results.
3562static Comparison getIntrinsicCmp(SelectionDAG &DAG, unsigned Opcode,
3563 SDValue Call, unsigned CCValid, uint64_t CC,
3565 Comparison C(Call, SDValue(), SDValue());
3566 C.Opcode = Opcode;
3567 C.CCValid = CCValid;
3568 if (Cond == ISD::SETEQ)
3569 // bit 3 for CC==0, bit 0 for CC==3, always false for CC>3.
3570 C.CCMask = CC < 4 ? 1 << (3 - CC) : 0;
3571 else if (Cond == ISD::SETNE)
3572 // ...and the inverse of that.
3573 C.CCMask = CC < 4 ? ~(1 << (3 - CC)) : -1;
3574 else if (Cond == ISD::SETLT || Cond == ISD::SETULT)
3575 // bits above bit 3 for CC==0 (always false), bits above bit 0 for CC==3,
3576 // always true for CC>3.
3577 C.CCMask = CC < 4 ? ~0U << (4 - CC) : -1;
3578 else if (Cond == ISD::SETGE || Cond == ISD::SETUGE)
3579 // ...and the inverse of that.
3580 C.CCMask = CC < 4 ? ~(~0U << (4 - CC)) : 0;
3581 else if (Cond == ISD::SETLE || Cond == ISD::SETULE)
3582 // bit 3 and above for CC==0, bit 0 and above for CC==3 (always true),
3583 // always true for CC>3.
3584 C.CCMask = CC < 4 ? ~0U << (3 - CC) : -1;
3585 else if (Cond == ISD::SETGT || Cond == ISD::SETUGT)
3586 // ...and the inverse of that.
3587 C.CCMask = CC < 4 ? ~(~0U << (3 - CC)) : 0;
3588 else
3589 llvm_unreachable("Unexpected integer comparison type");
3590 C.CCMask &= CCValid;
3591 return C;
3592}
3593
3594// Decide how to implement a comparison of type Cond between CmpOp0 with CmpOp1.
3595static Comparison getCmp(SelectionDAG &DAG, SDValue CmpOp0, SDValue CmpOp1,
3596 ISD::CondCode Cond, const SDLoc &DL,
3597 SDValue Chain = SDValue(),
3598 bool IsSignaling = false) {
3599 if (CmpOp1.getOpcode() == ISD::Constant) {
3600 assert(!Chain);
3601 unsigned Opcode, CCValid;
3602 if (CmpOp0.getOpcode() == ISD::INTRINSIC_W_CHAIN &&
3603 CmpOp0.getResNo() == 0 && CmpOp0->hasNUsesOfValue(1, 0) &&
3604 isIntrinsicWithCCAndChain(CmpOp0, Opcode, CCValid))
3605 return getIntrinsicCmp(DAG, Opcode, CmpOp0, CCValid,
3606 CmpOp1->getAsZExtVal(), Cond);
3607 if (CmpOp0.getOpcode() == ISD::INTRINSIC_WO_CHAIN &&
3608 CmpOp0.getResNo() == CmpOp0->getNumValues() - 1 &&
3609 isIntrinsicWithCC(CmpOp0, Opcode, CCValid))
3610 return getIntrinsicCmp(DAG, Opcode, CmpOp0, CCValid,
3611 CmpOp1->getAsZExtVal(), Cond);
3612 }
3613 Comparison C(CmpOp0, CmpOp1, Chain);
3614 C.CCMask = CCMaskForCondCode(Cond);
3615 if (C.Op0.getValueType().isFloatingPoint()) {
3616 C.CCValid = SystemZ::CCMASK_FCMP;
3617 if (!C.Chain)
3618 C.Opcode = SystemZISD::FCMP;
3619 else if (!IsSignaling)
3620 C.Opcode = SystemZISD::STRICT_FCMP;
3621 else
3622 C.Opcode = SystemZISD::STRICT_FCMPS;
3624 } else {
3625 assert(!C.Chain);
3626 C.CCValid = SystemZ::CCMASK_ICMP;
3627 C.Opcode = SystemZISD::ICMP;
3628 // Choose the type of comparison. Equality and inequality tests can
3629 // use either signed or unsigned comparisons. The choice also doesn't
3630 // matter if both sign bits are known to be clear. In those cases we
3631 // want to give the main isel code the freedom to choose whichever
3632 // form fits best.
3633 if (C.CCMask == SystemZ::CCMASK_CMP_EQ ||
3634 C.CCMask == SystemZ::CCMASK_CMP_NE ||
3635 (DAG.SignBitIsZero(C.Op0) && DAG.SignBitIsZero(C.Op1)))
3636 C.ICmpType = SystemZICMP::Any;
3637 else if (C.CCMask & SystemZ::CCMASK_CMP_UO)
3638 C.ICmpType = SystemZICMP::UnsignedOnly;
3639 else
3640 C.ICmpType = SystemZICMP::SignedOnly;
3641 C.CCMask &= ~SystemZ::CCMASK_CMP_UO;
3642 adjustForRedundantAnd(DAG, DL, C);
3643 adjustZeroCmp(DAG, DL, C);
3644 adjustSubwordCmp(DAG, DL, C);
3645 adjustForSubtraction(DAG, DL, C);
3647 adjustICmpTruncate(DAG, DL, C);
3648 }
3649
3650 if (shouldSwapCmpOperands(C)) {
3651 std::swap(C.Op0, C.Op1);
3652 C.CCMask = SystemZ::reverseCCMask(C.CCMask);
3653 }
3654
3656 adjustICmp128(DAG, DL, C);
3658 return C;
3659}
3660
3661// Emit the comparison instruction described by C.
3662static SDValue emitCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C) {
3663 if (!C.Op1.getNode()) {
3664 if (C.Opcode == SystemZISD::CMP_STACKGUARD)
3665 return DAG.getNode(SystemZISD::CMP_STACKGUARD, DL, MVT::i32, C.Op0);
3666 SDNode *Node;
3667 switch (C.Op0.getOpcode()) {
3669 Node = emitIntrinsicWithCCAndChain(DAG, C.Op0, C.Opcode);
3670 return SDValue(Node, 0);
3672 Node = emitIntrinsicWithCC(DAG, C.Op0, C.Opcode);
3673 return SDValue(Node, Node->getNumValues() - 1);
3674 default:
3675 llvm_unreachable("Invalid comparison operands");
3676 }
3677 }
3678 if (C.Opcode == SystemZISD::ICMP)
3679 return DAG.getNode(SystemZISD::ICMP, DL, MVT::i32, C.Op0, C.Op1,
3680 DAG.getTargetConstant(C.ICmpType, DL, MVT::i32));
3681 if (C.Opcode == SystemZISD::TM) {
3682 bool RegisterOnly = (bool(C.CCMask & SystemZ::CCMASK_TM_MIXED_MSB_0) !=
3684 return DAG.getNode(SystemZISD::TM, DL, MVT::i32, C.Op0, C.Op1,
3685 DAG.getTargetConstant(RegisterOnly, DL, MVT::i32));
3686 }
3687 if (C.Opcode == SystemZISD::VICMPES ||
3688 C.Opcode == SystemZISD::VICMPHS ||
3689 C.Opcode == SystemZISD::VICMPHLS ||
3690 C.Opcode == SystemZISD::VFCMPES ||
3691 C.Opcode == SystemZISD::VFCMPHS ||
3692 C.Opcode == SystemZISD::VFCMPHES) {
3693 EVT IntVT = C.Op0.getValueType().changeVectorElementTypeToInteger();
3694 SDVTList VTs = DAG.getVTList(IntVT, MVT::i32);
3695 SDValue Val = DAG.getNode(C.Opcode, DL, VTs, C.Op0, C.Op1);
3696 return SDValue(Val.getNode(), 1);
3697 }
3698 if (C.Chain) {
3699 SDVTList VTs = DAG.getVTList(MVT::i32, MVT::Other);
3700 return DAG.getNode(C.Opcode, DL, VTs, C.Chain, C.Op0, C.Op1);
3701 }
3702 return DAG.getNode(C.Opcode, DL, MVT::i32, C.Op0, C.Op1);
3703}
3704
3705// Implement a 32-bit *MUL_LOHI operation by extending both operands to
3706// 64 bits. Extend is the extension type to use. Store the high part
3707// in Hi and the low part in Lo.
3708static void lowerMUL_LOHI32(SelectionDAG &DAG, const SDLoc &DL, unsigned Extend,
3709 SDValue Op0, SDValue Op1, SDValue &Hi,
3710 SDValue &Lo) {
3711 Op0 = DAG.getNode(Extend, DL, MVT::i64, Op0);
3712 Op1 = DAG.getNode(Extend, DL, MVT::i64, Op1);
3713 SDValue Mul = DAG.getNode(ISD::MUL, DL, MVT::i64, Op0, Op1);
3714 Hi = DAG.getNode(ISD::SRL, DL, MVT::i64, Mul,
3715 DAG.getConstant(32, DL, MVT::i64));
3716 Hi = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Hi);
3717 Lo = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Mul);
3718}
3719
3720// Lower a binary operation that produces two VT results, one in each
3721// half of a GR128 pair. Op0 and Op1 are the VT operands to the operation,
3722// and Opcode performs the GR128 operation. Store the even register result
3723// in Even and the odd register result in Odd.
3724static void lowerGR128Binary(SelectionDAG &DAG, const SDLoc &DL, EVT VT,
3725 unsigned Opcode, SDValue Op0, SDValue Op1,
3726 SDValue &Even, SDValue &Odd) {
3727 SDValue Result = DAG.getNode(Opcode, DL, MVT::Untyped, Op0, Op1);
3728 bool Is32Bit = is32Bit(VT);
3729 Even = DAG.getTargetExtractSubreg(SystemZ::even128(Is32Bit), DL, VT, Result);
3730 Odd = DAG.getTargetExtractSubreg(SystemZ::odd128(Is32Bit), DL, VT, Result);
3731}
3732
3733// Return an i32 value that is 1 if the CC value produced by CCReg is
3734// in the mask CCMask and 0 otherwise. CC is known to have a value
3735// in CCValid, so other values can be ignored.
3736static SDValue emitSETCC(SelectionDAG &DAG, const SDLoc &DL, SDValue CCReg,
3737 unsigned CCValid, unsigned CCMask) {
3738 SDValue Ops[] = {DAG.getConstant(1, DL, MVT::i32),
3739 DAG.getConstant(0, DL, MVT::i32),
3740 DAG.getTargetConstant(CCValid, DL, MVT::i32),
3741 DAG.getTargetConstant(CCMask, DL, MVT::i32), CCReg};
3742 return DAG.getNode(SystemZISD::SELECT_CCMASK, DL, MVT::i32, Ops);
3743}
3744
3745// Return the SystemISD vector comparison operation for CC, or 0 if it cannot
3746// be done directly. Mode is CmpMode::Int for integer comparisons, CmpMode::FP
3747// for regular floating-point comparisons, CmpMode::StrictFP for strict (quiet)
3748// floating-point comparisons, and CmpMode::SignalingFP for strict signaling
3749// floating-point comparisons.
3752 switch (CC) {
3753 case ISD::SETOEQ:
3754 case ISD::SETEQ:
3755 switch (Mode) {
3756 case CmpMode::Int: return SystemZISD::VICMPE;
3757 case CmpMode::FP: return SystemZISD::VFCMPE;
3758 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPE;
3759 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPES;
3760 }
3761 llvm_unreachable("Bad mode");
3762
3763 case ISD::SETOGE:
3764 case ISD::SETGE:
3765 switch (Mode) {
3766 case CmpMode::Int: return 0;
3767 case CmpMode::FP: return SystemZISD::VFCMPHE;
3768 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPHE;
3769 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPHES;
3770 }
3771 llvm_unreachable("Bad mode");
3772
3773 case ISD::SETOGT:
3774 case ISD::SETGT:
3775 switch (Mode) {
3776 case CmpMode::Int: return SystemZISD::VICMPH;
3777 case CmpMode::FP: return SystemZISD::VFCMPH;
3778 case CmpMode::StrictFP: return SystemZISD::STRICT_VFCMPH;
3779 case CmpMode::SignalingFP: return SystemZISD::STRICT_VFCMPHS;
3780 }
3781 llvm_unreachable("Bad mode");
3782
3783 case ISD::SETUGT:
3784 switch (Mode) {
3785 case CmpMode::Int: return SystemZISD::VICMPHL;
3786 case CmpMode::FP: return 0;
3787 case CmpMode::StrictFP: return 0;
3788 case CmpMode::SignalingFP: return 0;
3789 }
3790 llvm_unreachable("Bad mode");
3791
3792 default:
3793 return 0;
3794 }
3795}
3796
3797// Return the SystemZISD vector comparison operation for CC or its inverse,
3798// or 0 if neither can be done directly. Indicate in Invert whether the
3799// result is for the inverse of CC. Mode is as above.
3801 bool &Invert) {
3802 if (unsigned Opcode = getVectorComparison(CC, Mode)) {
3803 Invert = false;
3804 return Opcode;
3805 }
3806
3807 CC = ISD::getSetCCInverse(CC, Mode == CmpMode::Int ? MVT::i32 : MVT::f32);
3808 if (unsigned Opcode = getVectorComparison(CC, Mode)) {
3809 Invert = true;
3810 return Opcode;
3811 }
3812
3813 return 0;
3814}
3815
3816// Return a v2f64 that contains the extended form of elements Start and Start+1
3817// of v4f32 value Op. If Chain is nonnull, return the strict form.
3818static SDValue expandV4F32ToV2F64(SelectionDAG &DAG, int Start, const SDLoc &DL,
3819 SDValue Op, SDValue Chain) {
3820 int Mask[] = { Start, -1, Start + 1, -1 };
3821 Op = DAG.getVectorShuffle(MVT::v4f32, DL, Op, DAG.getUNDEF(MVT::v4f32), Mask);
3822 if (Chain) {
3823 SDVTList VTs = DAG.getVTList(MVT::v2f64, MVT::Other);
3824 return DAG.getNode(SystemZISD::STRICT_VEXTEND, DL, VTs, Chain, Op);
3825 }
3826 return DAG.getNode(SystemZISD::VEXTEND, DL, MVT::v2f64, Op);
3827}
3828
3829// Build a comparison of vectors CmpOp0 and CmpOp1 using opcode Opcode,
3830// producing a result of type VT. If Chain is nonnull, return the strict form.
3831SDValue SystemZTargetLowering::getVectorCmp(SelectionDAG &DAG, unsigned Opcode,
3832 const SDLoc &DL, EVT VT,
3833 SDValue CmpOp0,
3834 SDValue CmpOp1,
3835 SDValue Chain) const {
3836 // There is no hardware support for v4f32 (unless we have the vector
3837 // enhancements facility 1), so extend the vector into two v2f64s
3838 // and compare those.
3839 if (CmpOp0.getValueType() == MVT::v4f32 &&
3840 !Subtarget.hasVectorEnhancements1()) {
3841 SDValue H0 = expandV4F32ToV2F64(DAG, 0, DL, CmpOp0, Chain);
3842 SDValue L0 = expandV4F32ToV2F64(DAG, 2, DL, CmpOp0, Chain);
3843 SDValue H1 = expandV4F32ToV2F64(DAG, 0, DL, CmpOp1, Chain);
3844 SDValue L1 = expandV4F32ToV2F64(DAG, 2, DL, CmpOp1, Chain);
3845 if (Chain) {
3846 SDVTList VTs = DAG.getVTList(MVT::v2i64, MVT::Other);
3847 SDValue HRes = DAG.getNode(Opcode, DL, VTs, Chain, H0, H1);
3848 SDValue LRes = DAG.getNode(Opcode, DL, VTs, Chain, L0, L1);
3849 SDValue Res = DAG.getNode(SystemZISD::PACK, DL, VT, HRes, LRes);
3850 SDValue Chains[6] = { H0.getValue(1), L0.getValue(1),
3851 H1.getValue(1), L1.getValue(1),
3852 HRes.getValue(1), LRes.getValue(1) };
3853 SDValue NewChain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chains);
3854 SDValue Ops[2] = { Res, NewChain };
3855 return DAG.getMergeValues(Ops, DL);
3856 }
3857 SDValue HRes = DAG.getNode(Opcode, DL, MVT::v2i64, H0, H1);
3858 SDValue LRes = DAG.getNode(Opcode, DL, MVT::v2i64, L0, L1);
3859 return DAG.getNode(SystemZISD::PACK, DL, VT, HRes, LRes);
3860 }
3861 if (Chain) {
3862 SDVTList VTs = DAG.getVTList(VT, MVT::Other);
3863 return DAG.getNode(Opcode, DL, VTs, Chain, CmpOp0, CmpOp1);
3864 }
3865 return DAG.getNode(Opcode, DL, VT, CmpOp0, CmpOp1);
3866}
3867
3868// Lower a vector comparison of type CC between CmpOp0 and CmpOp1, producing
3869// an integer mask of type VT. If Chain is nonnull, we have a strict
3870// floating-point comparison. If in addition IsSignaling is true, we have
3871// a strict signaling floating-point comparison.
3872SDValue SystemZTargetLowering::lowerVectorSETCC(SelectionDAG &DAG,
3873 const SDLoc &DL, EVT VT,
3874 ISD::CondCode CC,
3875 SDValue CmpOp0,
3876 SDValue CmpOp1,
3877 SDValue Chain,
3878 bool IsSignaling) const {
3879 bool IsFP = CmpOp0.getValueType().isFloatingPoint();
3880 assert (!Chain || IsFP);
3881 assert (!IsSignaling || Chain);
3882 CmpMode Mode = IsSignaling ? CmpMode::SignalingFP :
3883 Chain ? CmpMode::StrictFP : IsFP ? CmpMode::FP : CmpMode::Int;
3884 bool Invert = false;
3885 SDValue Cmp;
3886 switch (CC) {
3887 // Handle tests for order using (or (ogt y x) (oge x y)).
3888 case ISD::SETUO:
3889 Invert = true;
3890 [[fallthrough]];
3891 case ISD::SETO: {
3892 assert(IsFP && "Unexpected integer comparison");
3893 SDValue LT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3894 DL, VT, CmpOp1, CmpOp0, Chain);
3895 SDValue GE = getVectorCmp(DAG, getVectorComparison(ISD::SETOGE, Mode),
3896 DL, VT, CmpOp0, CmpOp1, Chain);
3897 Cmp = DAG.getNode(ISD::OR, DL, VT, LT, GE);
3898 if (Chain)
3899 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
3900 LT.getValue(1), GE.getValue(1));
3901 break;
3902 }
3903
3904 // Handle <> tests using (or (ogt y x) (ogt x y)).
3905 case ISD::SETUEQ:
3906 Invert = true;
3907 [[fallthrough]];
3908 case ISD::SETONE: {
3909 assert(IsFP && "Unexpected integer comparison");
3910 SDValue LT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3911 DL, VT, CmpOp1, CmpOp0, Chain);
3912 SDValue GT = getVectorCmp(DAG, getVectorComparison(ISD::SETOGT, Mode),
3913 DL, VT, CmpOp0, CmpOp1, Chain);
3914 Cmp = DAG.getNode(ISD::OR, DL, VT, LT, GT);
3915 if (Chain)
3916 Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other,
3917 LT.getValue(1), GT.getValue(1));
3918 break;
3919 }
3920
3921 // Otherwise a single comparison is enough. It doesn't really
3922 // matter whether we try the inversion or the swap first, since
3923 // there are no cases where both work.
3924 default:
3925 // Optimize sign-bit comparisons to signed compares.
3926 if (Mode == CmpMode::Int && (CC == ISD::SETEQ || CC == ISD::SETNE) &&
3928 unsigned EltSize = VT.getVectorElementType().getSizeInBits();
3929 APInt Mask;
3930 if (CmpOp0.getOpcode() == ISD::AND
3931 && ISD::isConstantSplatVector(CmpOp0.getOperand(1).getNode(), Mask)
3932 && Mask == APInt::getSignMask(EltSize)) {
3933 CC = CC == ISD::SETEQ ? ISD::SETGE : ISD::SETLT;
3934 CmpOp0 = CmpOp0.getOperand(0);
3935 }
3936 }
3937 if (unsigned Opcode = getVectorComparisonOrInvert(CC, Mode, Invert))
3938 Cmp = getVectorCmp(DAG, Opcode, DL, VT, CmpOp0, CmpOp1, Chain);
3939 else {
3941 if (unsigned Opcode = getVectorComparisonOrInvert(CC, Mode, Invert))
3942 Cmp = getVectorCmp(DAG, Opcode, DL, VT, CmpOp1, CmpOp0, Chain);
3943 else
3944 llvm_unreachable("Unhandled comparison");
3945 }
3946 if (Chain)
3947 Chain = Cmp.getValue(1);
3948 break;
3949 }
3950 if (Invert) {
3951 SDValue Mask =
3952 DAG.getSplatBuildVector(VT, DL, DAG.getAllOnesConstant(DL, MVT::i64));
3953 Cmp = DAG.getNode(ISD::XOR, DL, VT, Cmp, Mask);
3954 }
3955 if (Chain && Chain.getNode() != Cmp.getNode()) {
3956 SDValue Ops[2] = { Cmp, Chain };
3957 Cmp = DAG.getMergeValues(Ops, DL);
3958 }
3959 return Cmp;
3960}
3961
3962SDValue SystemZTargetLowering::lowerSETCC(SDValue Op,
3963 SelectionDAG &DAG) const {
3964 SDValue CmpOp0 = Op.getOperand(0);
3965 SDValue CmpOp1 = Op.getOperand(1);
3966 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(2))->get();
3967 SDLoc DL(Op);
3968 EVT VT = Op.getValueType();
3969 if (VT.isVector())
3970 return lowerVectorSETCC(DAG, DL, VT, CC, CmpOp0, CmpOp1);
3971
3972 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
3973 SDValue CCReg = emitCmp(DAG, DL, C);
3974 return emitSETCC(DAG, DL, CCReg, C.CCValid, C.CCMask);
3975}
3976
3977SDValue SystemZTargetLowering::lowerSTRICT_FSETCC(SDValue Op,
3978 SelectionDAG &DAG,
3979 bool IsSignaling) const {
3980 SDValue Chain = Op.getOperand(0);
3981 SDValue CmpOp0 = Op.getOperand(1);
3982 SDValue CmpOp1 = Op.getOperand(2);
3983 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(3))->get();
3984 SDLoc DL(Op);
3985 EVT VT = Op.getNode()->getValueType(0);
3986 if (VT.isVector()) {
3987 SDValue Res = lowerVectorSETCC(DAG, DL, VT, CC, CmpOp0, CmpOp1,
3988 Chain, IsSignaling);
3989 return Res.getValue(Op.getResNo());
3990 }
3991
3992 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL, Chain, IsSignaling));
3993 SDValue CCReg = emitCmp(DAG, DL, C);
3994 CCReg->setFlags(Op->getFlags());
3995 SDValue Result = emitSETCC(DAG, DL, CCReg, C.CCValid, C.CCMask);
3996 SDValue Ops[2] = { Result, CCReg.getValue(1) };
3997 return DAG.getMergeValues(Ops, DL);
3998}
3999
4000SDValue SystemZTargetLowering::lowerBR_CC(SDValue Op, SelectionDAG &DAG) const {
4001 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(1))->get();
4002 SDValue CmpOp0 = Op.getOperand(2);
4003 SDValue CmpOp1 = Op.getOperand(3);
4004 SDValue Dest = Op.getOperand(4);
4005 SDLoc DL(Op);
4006
4007 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
4008 SDValue CCReg = emitCmp(DAG, DL, C);
4009 return DAG.getNode(
4010 SystemZISD::BR_CCMASK, DL, Op.getValueType(), Op.getOperand(0),
4011 DAG.getTargetConstant(C.CCValid, DL, MVT::i32),
4012 DAG.getTargetConstant(C.CCMask, DL, MVT::i32), Dest, CCReg);
4013}
4014
4015// Return true if Pos is CmpOp and Neg is the negative of CmpOp,
4016// allowing Pos and Neg to be wider than CmpOp.
4017static bool isAbsolute(SDValue CmpOp, SDValue Pos, SDValue Neg) {
4018 return (Neg.getOpcode() == ISD::SUB &&
4019 Neg.getOperand(0).getOpcode() == ISD::Constant &&
4020 Neg.getConstantOperandVal(0) == 0 && Neg.getOperand(1) == Pos &&
4021 (Pos == CmpOp || (Pos.getOpcode() == ISD::SIGN_EXTEND &&
4022 Pos.getOperand(0) == CmpOp)));
4023}
4024
4025// Return the absolute or negative absolute of Op; IsNegative decides which.
4027 bool IsNegative) {
4028 Op = DAG.getNode(ISD::ABS, DL, Op.getValueType(), Op);
4029 if (IsNegative)
4030 Op = DAG.getNode(ISD::SUB, DL, Op.getValueType(),
4031 DAG.getConstant(0, DL, Op.getValueType()), Op);
4032 return Op;
4033}
4034
4036 Comparison C, SDValue TrueOp, SDValue FalseOp) {
4037 EVT VT = MVT::i128;
4038 unsigned Op;
4039
4040 if (C.CCMask == SystemZ::CCMASK_CMP_NE ||
4041 C.CCMask == SystemZ::CCMASK_CMP_GE ||
4042 C.CCMask == SystemZ::CCMASK_CMP_LE) {
4043 std::swap(TrueOp, FalseOp);
4044 C.CCMask ^= C.CCValid;
4045 }
4046 if (C.CCMask == SystemZ::CCMASK_CMP_LT) {
4047 std::swap(C.Op0, C.Op1);
4048 C.CCMask = SystemZ::CCMASK_CMP_GT;
4049 }
4050 switch (C.CCMask) {
4052 Op = SystemZISD::VICMPE;
4053 break;
4055 if (C.ICmpType == SystemZICMP::UnsignedOnly)
4056 Op = SystemZISD::VICMPHL;
4057 else
4058 Op = SystemZISD::VICMPH;
4059 break;
4060 default:
4061 llvm_unreachable("Unhandled comparison");
4062 break;
4063 }
4064
4065 SDValue Mask = DAG.getNode(Op, DL, VT, C.Op0, C.Op1);
4066 TrueOp = DAG.getNode(ISD::AND, DL, VT, TrueOp, Mask);
4067 FalseOp = DAG.getNode(ISD::AND, DL, VT, FalseOp, DAG.getNOT(DL, Mask, VT));
4068 return DAG.getNode(ISD::OR, DL, VT, TrueOp, FalseOp);
4069}
4070
4071SDValue SystemZTargetLowering::lowerSELECT_CC(SDValue Op,
4072 SelectionDAG &DAG) const {
4073 SDValue CmpOp0 = Op.getOperand(0);
4074 SDValue CmpOp1 = Op.getOperand(1);
4075 SDValue TrueOp = Op.getOperand(2);
4076 SDValue FalseOp = Op.getOperand(3);
4077 ISD::CondCode CC = cast<CondCodeSDNode>(Op.getOperand(4))->get();
4078 SDLoc DL(Op);
4079
4080 // SELECT_CC involving f16 will not have the cmp-ops promoted by the
4081 // legalizer, as it will be handled according to the type of the resulting
4082 // value. Extend them here if needed.
4083 if (CmpOp0.getSimpleValueType() == MVT::f16) {
4084 CmpOp0 = DAG.getFPExtendOrRound(CmpOp0, SDLoc(CmpOp0), MVT::f32);
4085 CmpOp1 = DAG.getFPExtendOrRound(CmpOp1, SDLoc(CmpOp1), MVT::f32);
4086 }
4087
4088 Comparison C(getCmp(DAG, CmpOp0, CmpOp1, CC, DL));
4089
4090 // Check for absolute and negative-absolute selections, including those
4091 // where the comparison value is sign-extended (for LPGFR and LNGFR).
4092 // This check supplements the one in DAGCombiner.
4093 if (C.Opcode == SystemZISD::ICMP && C.CCMask != SystemZ::CCMASK_CMP_EQ &&
4094 C.CCMask != SystemZ::CCMASK_CMP_NE &&
4095 C.Op1.getOpcode() == ISD::Constant &&
4096 cast<ConstantSDNode>(C.Op1)->getValueSizeInBits(0) <= 64 &&
4097 C.Op1->getAsZExtVal() == 0) {
4098 if (isAbsolute(C.Op0, TrueOp, FalseOp))
4099 return getAbsolute(DAG, DL, TrueOp, C.CCMask & SystemZ::CCMASK_CMP_LT);
4100 if (isAbsolute(C.Op0, FalseOp, TrueOp))
4101 return getAbsolute(DAG, DL, FalseOp, C.CCMask & SystemZ::CCMASK_CMP_GT);
4102 }
4103
4104 if (Subtarget.hasVectorEnhancements3() &&
4105 C.Opcode == SystemZISD::ICMP &&
4106 C.Op0.getValueType() == MVT::i128 &&
4107 TrueOp.getValueType() == MVT::i128) {
4108 return getI128Select(DAG, DL, C, TrueOp, FalseOp);
4109 }
4110
4111 SDValue CCReg = emitCmp(DAG, DL, C);
4112 SDValue Ops[] = {TrueOp, FalseOp,
4113 DAG.getTargetConstant(C.CCValid, DL, MVT::i32),
4114 DAG.getTargetConstant(C.CCMask, DL, MVT::i32), CCReg};
4115
4116 return DAG.getNode(SystemZISD::SELECT_CCMASK, DL, Op.getValueType(), Ops);
4117}
4118
4119SDValue SystemZTargetLowering::lowerGlobalAddress(GlobalAddressSDNode *Node,
4120 SelectionDAG &DAG) const {
4121 SDLoc DL(Node);
4122 const GlobalValue *GV = Node->getGlobal();
4123 int64_t Offset = Node->getOffset();
4124 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4126
4127 SDValue Result;
4128 if (Subtarget.isPC32DBLSymbol(GV, CM)) {
4129 if (isInt<32>(Offset)) {
4130 // Assign anchors at 1<<12 byte boundaries.
4131 uint64_t Anchor = Offset & ~uint64_t(0xfff);
4132 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT, Anchor);
4133 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4134
4135 // The offset can be folded into the address if it is aligned to a
4136 // halfword.
4137 Offset -= Anchor;
4138 if (Offset != 0 && (Offset & 1) == 0) {
4139 SDValue Full =
4140 DAG.getTargetGlobalAddress(GV, DL, PtrVT, Anchor + Offset);
4141 Result = DAG.getNode(SystemZISD::PCREL_OFFSET, DL, PtrVT, Full, Result);
4142 Offset = 0;
4143 }
4144 } else {
4145 // Conservatively load a constant offset greater than 32 bits into a
4146 // register below.
4147 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT);
4148 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4149 }
4150 } else if (Subtarget.isTargetELF()) {
4151 Result = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0, SystemZII::MO_GOT);
4152 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4153 Result = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Result,
4155 } else if (Subtarget.isTargetzOS()) {
4156 Result = getADAEntry(DAG, GV, DL, PtrVT);
4157 } else
4158 llvm_unreachable("Unexpected Subtarget");
4159
4160 // If there was a non-zero offset that we didn't fold, create an explicit
4161 // addition for it.
4162 if (Offset != 0)
4163 Result = DAG.getNode(ISD::ADD, DL, PtrVT, Result,
4164 DAG.getSignedConstant(Offset, DL, PtrVT));
4165
4166 return Result;
4167}
4168
4169SDValue SystemZTargetLowering::lowerTLSGetOffset(GlobalAddressSDNode *Node,
4170 SelectionDAG &DAG,
4171 unsigned Opcode,
4172 SDValue GOTOffset) const {
4173 SDLoc DL(Node);
4174 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4175 SDValue Chain = DAG.getEntryNode();
4176 SDValue Glue;
4177
4180 report_fatal_error("In GHC calling convention TLS is not supported");
4181
4182 // __tls_get_offset takes the GOT offset in %r2 and the GOT in %r12.
4183 SDValue GOT = DAG.getGLOBAL_OFFSET_TABLE(PtrVT);
4184 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R12D, GOT, Glue);
4185 Glue = Chain.getValue(1);
4186 Chain = DAG.getCopyToReg(Chain, DL, SystemZ::R2D, GOTOffset, Glue);
4187 Glue = Chain.getValue(1);
4188
4189 // The first call operand is the chain and the second is the TLS symbol.
4191 Ops.push_back(Chain);
4192 Ops.push_back(DAG.getTargetGlobalAddress(Node->getGlobal(), DL,
4193 Node->getValueType(0),
4194 0, 0));
4195
4196 // Add argument registers to the end of the list so that they are
4197 // known live into the call.
4198 Ops.push_back(DAG.getRegister(SystemZ::R2D, PtrVT));
4199 Ops.push_back(DAG.getRegister(SystemZ::R12D, PtrVT));
4200
4201 // Add a register mask operand representing the call-preserved registers.
4202 const TargetRegisterInfo *TRI = Subtarget.getRegisterInfo();
4203 const uint32_t *Mask =
4204 TRI->getCallPreservedMask(DAG.getMachineFunction(), CallingConv::C);
4205 assert(Mask && "Missing call preserved mask for calling convention");
4206 Ops.push_back(DAG.getRegisterMask(Mask));
4207
4208 // Glue the call to the argument copies.
4209 Ops.push_back(Glue);
4210
4211 // Emit the call.
4212 SDVTList NodeTys = DAG.getVTList(MVT::Other, MVT::Glue);
4213 Chain = DAG.getNode(Opcode, DL, NodeTys, Ops);
4214 Glue = Chain.getValue(1);
4215
4216 // Copy the return value from %r2.
4217 return DAG.getCopyFromReg(Chain, DL, SystemZ::R2D, PtrVT, Glue);
4218}
4219
4220SDValue SystemZTargetLowering::lowerThreadPointer(const SDLoc &DL,
4221 SelectionDAG &DAG) const {
4222 SDValue Chain = DAG.getEntryNode();
4223 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4224
4225 // The high part of the thread pointer is in access register 0.
4226 SDValue TPHi = DAG.getCopyFromReg(Chain, DL, SystemZ::A0, MVT::i32);
4227 TPHi = DAG.getNode(ISD::ANY_EXTEND, DL, PtrVT, TPHi);
4228
4229 // The low part of the thread pointer is in access register 1.
4230 SDValue TPLo = DAG.getCopyFromReg(Chain, DL, SystemZ::A1, MVT::i32);
4231 TPLo = DAG.getNode(ISD::ZERO_EXTEND, DL, PtrVT, TPLo);
4232
4233 // Merge them into a single 64-bit address.
4234 SDValue TPHiShifted = DAG.getNode(ISD::SHL, DL, PtrVT, TPHi,
4235 DAG.getConstant(32, DL, PtrVT));
4236 return DAG.getNode(ISD::OR, DL, PtrVT, TPHiShifted, TPLo);
4237}
4238
4239SDValue SystemZTargetLowering::lowerGlobalTLSAddress(GlobalAddressSDNode *Node,
4240 SelectionDAG &DAG) const {
4241 if (DAG.getTarget().useEmulatedTLS())
4242 return LowerToTLSEmulatedModel(Node, DAG);
4243 SDLoc DL(Node);
4244 const GlobalValue *GV = Node->getGlobal();
4245 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4246 TLSModel::Model model = DAG.getTarget().getTLSModel(GV);
4247
4250 report_fatal_error("In GHC calling convention TLS is not supported");
4251
4252 SDValue TP = lowerThreadPointer(DL, DAG);
4253
4254 // Get the offset of GA from the thread pointer, based on the TLS model.
4255 SDValue Offset;
4256 switch (model) {
4258 // Load the GOT offset of the tls_index (module ID / per-symbol offset).
4259 SystemZConstantPoolValue *CPV =
4261
4262 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4263 Offset = DAG.getLoad(
4264 PtrVT, DL, DAG.getEntryNode(), Offset,
4266
4267 // Call __tls_get_offset to retrieve the offset.
4268 Offset = lowerTLSGetOffset(Node, DAG, SystemZISD::TLS_GDCALL, Offset);
4269 break;
4270 }
4271
4273 // Load the GOT offset of the module ID.
4274 SystemZConstantPoolValue *CPV =
4276
4277 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4278 Offset = DAG.getLoad(
4279 PtrVT, DL, DAG.getEntryNode(), Offset,
4281
4282 // Call __tls_get_offset to retrieve the module base offset.
4283 Offset = lowerTLSGetOffset(Node, DAG, SystemZISD::TLS_LDCALL, Offset);
4284
4285 // Note: The SystemZLDCleanupPass will remove redundant computations
4286 // of the module base offset. Count total number of local-dynamic
4287 // accesses to trigger execution of that pass.
4288 SystemZMachineFunctionInfo* MFI =
4289 DAG.getMachineFunction().getInfo<SystemZMachineFunctionInfo>();
4291
4292 // Add the per-symbol offset.
4294
4295 SDValue DTPOffset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4296 DTPOffset = DAG.getLoad(
4297 PtrVT, DL, DAG.getEntryNode(), DTPOffset,
4299
4300 Offset = DAG.getNode(ISD::ADD, DL, PtrVT, Offset, DTPOffset);
4301 break;
4302 }
4303
4304 case TLSModel::InitialExec: {
4305 // Load the offset from the GOT.
4306 Offset = DAG.getTargetGlobalAddress(GV, DL, PtrVT, 0,
4308 Offset = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Offset);
4309 Offset =
4310 DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Offset,
4312 break;
4313 }
4314
4315 case TLSModel::LocalExec: {
4316 // Force the offset into the constant pool and load it from there.
4317 SystemZConstantPoolValue *CPV =
4319
4320 Offset = DAG.getConstantPool(CPV, PtrVT, Align(8));
4321 Offset = DAG.getLoad(
4322 PtrVT, DL, DAG.getEntryNode(), Offset,
4324 break;
4325 }
4326 }
4327
4328 // Add the base and offset together.
4329 return DAG.getNode(ISD::ADD, DL, PtrVT, TP, Offset);
4330}
4331
4332SDValue SystemZTargetLowering::lowerBlockAddress(BlockAddressSDNode *Node,
4333 SelectionDAG &DAG) const {
4334 SDLoc DL(Node);
4335 const BlockAddress *BA = Node->getBlockAddress();
4336 int64_t Offset = Node->getOffset();
4337 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4338
4339 SDValue Result = DAG.getTargetBlockAddress(BA, PtrVT, Offset);
4340 Result = DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4341 return Result;
4342}
4343
4344SDValue SystemZTargetLowering::lowerJumpTable(JumpTableSDNode *JT,
4345 SelectionDAG &DAG) const {
4346 SDLoc DL(JT);
4347 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4348 SDValue Result = DAG.getTargetJumpTable(JT->getIndex(), PtrVT);
4349
4350 // Use LARL to load the address of the table.
4351 return DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4352}
4353
4354SDValue SystemZTargetLowering::lowerConstantPool(ConstantPoolSDNode *CP,
4355 SelectionDAG &DAG) const {
4356 SDLoc DL(CP);
4357 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4358
4359 SDValue Result;
4361 Result =
4362 DAG.getTargetConstantPool(CP->getMachineCPVal(), PtrVT, CP->getAlign());
4363 else
4364 Result = DAG.getTargetConstantPool(CP->getConstVal(), PtrVT, CP->getAlign(),
4365 CP->getOffset());
4366
4367 // Use LARL to load the address of the constant pool entry.
4368 return DAG.getNode(SystemZISD::PCREL_WRAPPER, DL, PtrVT, Result);
4369}
4370
4371SDValue SystemZTargetLowering::lowerFRAMEADDR(SDValue Op,
4372 SelectionDAG &DAG) const {
4373 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
4375 MachineFrameInfo &MFI = MF.getFrameInfo();
4376 MFI.setFrameAddressIsTaken(true);
4377
4378 SDLoc DL(Op);
4379 unsigned Depth = Op.getConstantOperandVal(0);
4380 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4381
4382 // By definition, the frame address is the address of the back chain. (In
4383 // the case of packed stack without backchain, return the address where the
4384 // backchain would have been stored. This will either be an unused space or
4385 // contain a saved register).
4386 int BackChainIdx = TFL->getOrCreateFramePointerSaveIndex(MF);
4387 SDValue BackChain = DAG.getFrameIndex(BackChainIdx, PtrVT);
4388
4389 if (Depth > 0) {
4390 // FIXME The frontend should detect this case.
4391 if (!MF.getSubtarget<SystemZSubtarget>().hasBackChain())
4392 report_fatal_error("Unsupported stack frame traversal count");
4393
4394 SDValue Offset = DAG.getConstant(TFL->getBackchainOffset(MF), DL, PtrVT);
4395 while (Depth--) {
4396 BackChain = DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), BackChain,
4397 MachinePointerInfo());
4398 BackChain = DAG.getNode(ISD::ADD, DL, PtrVT, BackChain, Offset);
4399 }
4400 }
4401
4402 return BackChain;
4403}
4404
4405SDValue SystemZTargetLowering::lowerRETURNADDR(SDValue Op,
4406 SelectionDAG &DAG) const {
4408 MachineFrameInfo &MFI = MF.getFrameInfo();
4409 MFI.setReturnAddressIsTaken(true);
4410
4411 SDLoc DL(Op);
4412 unsigned Depth = Op.getConstantOperandVal(0);
4413 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4414
4415 if (Depth > 0) {
4416 // FIXME The frontend should detect this case.
4417 if (!MF.getSubtarget<SystemZSubtarget>().hasBackChain())
4418 report_fatal_error("Unsupported stack frame traversal count");
4419
4420 SDValue FrameAddr = lowerFRAMEADDR(Op, DAG);
4421 const auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
4422 int Offset = TFL->getReturnAddressOffset(MF);
4423 SDValue Ptr = DAG.getNode(ISD::ADD, DL, PtrVT, FrameAddr,
4424 DAG.getSignedConstant(Offset, DL, PtrVT));
4425 return DAG.getLoad(PtrVT, DL, DAG.getEntryNode(), Ptr,
4426 MachinePointerInfo());
4427 }
4428
4429 // Return R14D (Elf) / R7D (XPLINK), which has the return address. Mark it an
4430 // implicit live-in.
4431 SystemZCallingConventionRegisters *CCR = Subtarget.getSpecialRegisters();
4433 &SystemZ::GR64BitRegClass);
4434 return DAG.getCopyFromReg(DAG.getEntryNode(), DL, LinkReg, PtrVT);
4435}
4436
4437SDValue SystemZTargetLowering::lowerBITCAST(SDValue Op,
4438 SelectionDAG &DAG) const {
4439 SDLoc DL(Op);
4440 SDValue In = Op.getOperand(0);
4441 EVT InVT = In.getValueType();
4442 EVT ResVT = Op.getValueType();
4443
4444 // Convert loads directly. This is normally done by DAGCombiner,
4445 // but we need this case for bitcasts that are created during lowering
4446 // and which are then lowered themselves.
4447 if (auto *LoadN = dyn_cast<LoadSDNode>(In))
4448 if (ISD::isNormalLoad(LoadN)) {
4449 SDValue NewLoad = DAG.getLoad(ResVT, DL, LoadN->getChain(),
4450 LoadN->getBasePtr(), LoadN->getMemOperand());
4451 // Update the chain uses.
4452 DAG.ReplaceAllUsesOfValueWith(SDValue(LoadN, 1), NewLoad.getValue(1));
4453 return NewLoad;
4454 }
4455
4456 if (InVT == MVT::i32 && ResVT == MVT::f32) {
4457 SDValue In64;
4458 if (Subtarget.hasHighWord()) {
4459 SDNode *U64 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL,
4460 MVT::i64);
4461 In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h32, DL,
4462 MVT::i64, SDValue(U64, 0), In);
4463 } else {
4464 In64 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, In);
4465 In64 = DAG.getNode(ISD::SHL, DL, MVT::i64, In64,
4466 DAG.getConstant(32, DL, MVT::i64));
4467 }
4468 SDValue Out64 = DAG.getNode(ISD::BITCAST, DL, MVT::f64, In64);
4469 return DAG.getTargetExtractSubreg(SystemZ::subreg_h32,
4470 DL, MVT::f32, Out64);
4471 }
4472 if (InVT == MVT::f32 && ResVT == MVT::i32) {
4473 SDNode *U64 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f64);
4474 SDValue In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h32, DL,
4475 MVT::f64, SDValue(U64, 0), In);
4476 SDValue Out64 = DAG.getNode(ISD::BITCAST, DL, MVT::i64, In64);
4477 if (Subtarget.hasHighWord())
4478 return DAG.getTargetExtractSubreg(SystemZ::subreg_h32, DL,
4479 MVT::i32, Out64);
4480 SDValue Shift = DAG.getNode(ISD::SRL, DL, MVT::i64, Out64,
4481 DAG.getConstant(32, DL, MVT::i64));
4482 return DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Shift);
4483 }
4484 llvm_unreachable("Unexpected bitcast combination");
4485}
4486
4487SDValue SystemZTargetLowering::lowerVASTART(SDValue Op,
4488 SelectionDAG &DAG) const {
4489
4490 if (Subtarget.isTargetXPLINK64())
4491 return lowerVASTART_XPLINK(Op, DAG);
4492 else
4493 return lowerVASTART_ELF(Op, DAG);
4494}
4495
4496SDValue SystemZTargetLowering::lowerVASTART_XPLINK(SDValue Op,
4497 SelectionDAG &DAG) const {
4499 SystemZMachineFunctionInfo *FuncInfo =
4500 MF.getInfo<SystemZMachineFunctionInfo>();
4501
4502 SDLoc DL(Op);
4503
4504 // vastart just stores the address of the VarArgsFrameIndex slot into the
4505 // memory location argument.
4506 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4507 SDValue FR = DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT);
4508 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4509 return DAG.getStore(Op.getOperand(0), DL, FR, Op.getOperand(1),
4510 MachinePointerInfo(SV));
4511}
4512
4513SDValue SystemZTargetLowering::lowerVASTART_ELF(SDValue Op,
4514 SelectionDAG &DAG) const {
4516 SystemZMachineFunctionInfo *FuncInfo =
4517 MF.getInfo<SystemZMachineFunctionInfo>();
4518 EVT PtrVT = getPointerTy(DAG.getDataLayout());
4519
4520 SDValue Chain = Op.getOperand(0);
4521 SDValue Addr = Op.getOperand(1);
4522 const Value *SV = cast<SrcValueSDNode>(Op.getOperand(2))->getValue();
4523 SDLoc DL(Op);
4524
4525 // The initial values of each field.
4526 const unsigned NumFields = 4;
4527 SDValue Fields[NumFields] = {
4528 DAG.getConstant(FuncInfo->getVarArgsFirstGPR(), DL, PtrVT),
4529 DAG.getConstant(FuncInfo->getVarArgsFirstFPR(), DL, PtrVT),
4530 DAG.getFrameIndex(FuncInfo->getVarArgsFrameIndex(), PtrVT),
4531 DAG.getFrameIndex(FuncInfo->getRegSaveFrameIndex(), PtrVT)
4532 };
4533
4534 // Store each field into its respective slot.
4535 SDValue MemOps[NumFields];
4536 unsigned Offset = 0;
4537 for (unsigned I = 0; I < NumFields; ++I) {
4538 SDValue FieldAddr = Addr;
4539 if (Offset != 0)
4540 FieldAddr = DAG.getNode(ISD::ADD, DL, PtrVT, FieldAddr,
4542 MemOps[I] = DAG.getStore(Chain, DL, Fields[I], FieldAddr,
4543 MachinePointerInfo(SV, Offset));
4544 Offset += 8;
4545 }
4546 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, MemOps);
4547}
4548
4549SDValue SystemZTargetLowering::lowerVACOPY(SDValue Op,
4550 SelectionDAG &DAG) const {
4551 SDValue Chain = Op.getOperand(0);
4552 SDValue DstPtr = Op.getOperand(1);
4553 SDValue SrcPtr = Op.getOperand(2);
4554 const Value *DstSV = cast<SrcValueSDNode>(Op.getOperand(3))->getValue();
4555 const Value *SrcSV = cast<SrcValueSDNode>(Op.getOperand(4))->getValue();
4556 SDLoc DL(Op);
4557
4558 uint32_t Sz =
4559 Subtarget.isTargetXPLINK64() ? DAG.getDataLayout().getPointerSize(0) : 32;
4560 return DAG.getMemcpy(Chain, DL, DstPtr, SrcPtr, DAG.getIntPtrConstant(Sz, DL),
4561 Align(8), Align(8), /*isVolatile*/ false,
4562 /*AlwaysInline*/ false,
4563 /*CI=*/nullptr, std::nullopt, MachinePointerInfo(DstSV),
4564 MachinePointerInfo(SrcSV));
4565}
4566
4567SDValue
4568SystemZTargetLowering::lowerDYNAMIC_STACKALLOC(SDValue Op,
4569 SelectionDAG &DAG) const {
4570 if (Subtarget.isTargetXPLINK64())
4571 return lowerDYNAMIC_STACKALLOC_XPLINK(Op, DAG);
4572 else
4573 return lowerDYNAMIC_STACKALLOC_ELF(Op, DAG);
4574}
4575
4576SDValue
4577SystemZTargetLowering::lowerDYNAMIC_STACKALLOC_XPLINK(SDValue Op,
4578 SelectionDAG &DAG) const {
4579 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
4581 bool RealignOpt = !MF.getFunction().hasFnAttribute("no-realign-stack");
4582 SDValue Chain = Op.getOperand(0);
4583 SDValue Size = Op.getOperand(1);
4584 SDValue Align = Op.getOperand(2);
4585 SDLoc DL(Op);
4586
4587 // If user has set the no alignment function attribute, ignore
4588 // alloca alignments.
4589 uint64_t AlignVal = (RealignOpt ? Align->getAsZExtVal() : 0);
4590
4591 uint64_t StackAlign = TFI->getStackAlignment();
4592 uint64_t RequiredAlign = std::max(AlignVal, StackAlign);
4593 uint64_t ExtraAlignSpace = RequiredAlign - StackAlign;
4594
4595 SDValue NeededSpace = Size;
4596
4597 // Add extra space for alignment if needed.
4598 EVT PtrVT = getPointerTy(MF.getDataLayout());
4599 if (ExtraAlignSpace)
4600 NeededSpace = DAG.getNode(ISD::ADD, DL, PtrVT, NeededSpace,
4601 DAG.getConstant(ExtraAlignSpace, DL, PtrVT));
4602
4603 bool IsSigned = false;
4604 bool DoesNotReturn = false;
4605 bool IsReturnValueUsed = false;
4606 EVT VT = Op.getValueType();
4607 SDValue AllocaCall =
4608 makeExternalCall(Chain, DAG, "@@ALCAXP", VT, ArrayRef(NeededSpace),
4609 CallingConv::C, IsSigned, DL, DoesNotReturn,
4610 IsReturnValueUsed)
4611 .first;
4612
4613 // Perform a CopyFromReg from %GPR4 (stack pointer register). Chain and Glue
4614 // to end of call in order to ensure it isn't broken up from the call
4615 // sequence.
4616 auto &Regs = Subtarget.getSpecialRegisters<SystemZXPLINK64Registers>();
4617 Register SPReg = Regs.getStackPointerRegister();
4618 Chain = AllocaCall.getValue(1);
4619 SDValue Glue = AllocaCall.getValue(2);
4620 SDValue NewSPRegNode = DAG.getCopyFromReg(Chain, DL, SPReg, PtrVT, Glue);
4621 Chain = NewSPRegNode.getValue(1);
4622
4623 MVT PtrMVT = getPointerMemTy(MF.getDataLayout());
4624 SDValue ArgAdjust = DAG.getNode(SystemZISD::ADJDYNALLOC, DL, PtrMVT);
4625 SDValue Result = DAG.getNode(ISD::ADD, DL, PtrMVT, NewSPRegNode, ArgAdjust);
4626
4627 // Dynamically realign if needed.
4628 if (ExtraAlignSpace) {
4629 Result = DAG.getNode(ISD::ADD, DL, PtrVT, Result,
4630 DAG.getConstant(ExtraAlignSpace, DL, PtrVT));
4631 Result = DAG.getNode(ISD::AND, DL, PtrVT, Result,
4632 DAG.getConstant(~(RequiredAlign - 1), DL, PtrVT));
4633 }
4634
4635 SDValue Ops[2] = {Result, Chain};
4636 return DAG.getMergeValues(Ops, DL);
4637}
4638
4639SDValue
4640SystemZTargetLowering::lowerDYNAMIC_STACKALLOC_ELF(SDValue Op,
4641 SelectionDAG &DAG) const {
4642 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
4644 bool RealignOpt = !MF.getFunction().hasFnAttribute("no-realign-stack");
4645 bool StoreBackchain = MF.getSubtarget<SystemZSubtarget>().hasBackChain();
4646
4647 SDValue Chain = Op.getOperand(0);
4648 SDValue Size = Op.getOperand(1);
4649 SDValue Align = Op.getOperand(2);
4650 SDLoc DL(Op);
4651
4652 // If user has set the no alignment function attribute, ignore
4653 // alloca alignments.
4654 uint64_t AlignVal = (RealignOpt ? Align->getAsZExtVal() : 0);
4655
4656 uint64_t StackAlign = TFI->getStackAlignment();
4657 uint64_t RequiredAlign = std::max(AlignVal, StackAlign);
4658 uint64_t ExtraAlignSpace = RequiredAlign - StackAlign;
4659
4661 SDValue NeededSpace = Size;
4662
4663 // Get a reference to the stack pointer.
4664 SDValue OldSP = DAG.getCopyFromReg(Chain, DL, SPReg, MVT::i64);
4665
4666 // If we need a backchain, save it now.
4667 SDValue Backchain;
4668 if (StoreBackchain)
4669 Backchain = DAG.getLoad(MVT::i64, DL, Chain, getBackchainAddress(OldSP, DAG),
4670 MachinePointerInfo());
4671
4672 // Add extra space for alignment if needed.
4673 if (ExtraAlignSpace)
4674 NeededSpace = DAG.getNode(ISD::ADD, DL, MVT::i64, NeededSpace,
4675 DAG.getConstant(ExtraAlignSpace, DL, MVT::i64));
4676
4677 // Get the new stack pointer value.
4678 SDValue NewSP;
4679 if (hasInlineStackProbe(MF)) {
4680 NewSP = DAG.getNode(SystemZISD::PROBED_ALLOCA, DL,
4681 DAG.getVTList(MVT::i64, MVT::Other), Chain, OldSP, NeededSpace);
4682 Chain = NewSP.getValue(1);
4683 }
4684 else {
4685 NewSP = DAG.getNode(ISD::SUB, DL, MVT::i64, OldSP, NeededSpace);
4686 // Copy the new stack pointer back.
4687 Chain = DAG.getCopyToReg(Chain, DL, SPReg, NewSP);
4688 }
4689
4690 // The allocated data lives above the 160 bytes allocated for the standard
4691 // frame, plus any outgoing stack arguments. We don't know how much that
4692 // amounts to yet, so emit a special ADJDYNALLOC placeholder.
4693 SDValue ArgAdjust = DAG.getNode(SystemZISD::ADJDYNALLOC, DL, MVT::i64);
4694 SDValue Result = DAG.getNode(ISD::ADD, DL, MVT::i64, NewSP, ArgAdjust);
4695
4696 // Dynamically realign if needed.
4697 if (RequiredAlign > StackAlign) {
4698 Result =
4699 DAG.getNode(ISD::ADD, DL, MVT::i64, Result,
4700 DAG.getConstant(ExtraAlignSpace, DL, MVT::i64));
4701 Result =
4702 DAG.getNode(ISD::AND, DL, MVT::i64, Result,
4703 DAG.getConstant(~(RequiredAlign - 1), DL, MVT::i64));
4704 }
4705
4706 if (StoreBackchain)
4707 Chain = DAG.getStore(Chain, DL, Backchain, getBackchainAddress(NewSP, DAG),
4708 MachinePointerInfo());
4709
4710 SDValue Ops[2] = { Result, Chain };
4711 return DAG.getMergeValues(Ops, DL);
4712}
4713
4714SDValue SystemZTargetLowering::lowerGET_DYNAMIC_AREA_OFFSET(
4715 SDValue Op, SelectionDAG &DAG) const {
4716 SDLoc DL(Op);
4717
4718 return DAG.getNode(SystemZISD::ADJDYNALLOC, DL, MVT::i64);
4719}
4720
4721SDValue SystemZTargetLowering::lowerMULH(SDValue Op,
4722 SelectionDAG &DAG,
4723 unsigned Opcode) const {
4724 EVT VT = Op.getValueType();
4725 SDLoc DL(Op);
4726 SDValue Even, Odd;
4727
4728 // This custom expander is only used on z17 and later for 64-bit types.
4729 assert(!is32Bit(VT));
4730 assert(Subtarget.hasMiscellaneousExtensions2());
4731
4732 // SystemZISD::xMUL_LOHI returns the low result in the odd register and
4733 // the high result in the even register. Return the latter.
4734 lowerGR128Binary(DAG, DL, VT, Opcode,
4735 Op.getOperand(0), Op.getOperand(1), Even, Odd);
4736 return Even;
4737}
4738
4739SDValue SystemZTargetLowering::lowerSMUL_LOHI(SDValue Op,
4740 SelectionDAG &DAG) const {
4741 EVT VT = Op.getValueType();
4742 SDLoc DL(Op);
4743 SDValue Ops[2];
4744 if (is32Bit(VT))
4745 // Just do a normal 64-bit multiplication and extract the results.
4746 // We define this so that it can be used for constant division.
4747 lowerMUL_LOHI32(DAG, DL, ISD::SIGN_EXTEND, Op.getOperand(0),
4748 Op.getOperand(1), Ops[1], Ops[0]);
4749 else if (Subtarget.hasMiscellaneousExtensions2())
4750 // SystemZISD::SMUL_LOHI returns the low result in the odd register and
4751 // the high result in the even register. ISD::SMUL_LOHI is defined to
4752 // return the low half first, so the results are in reverse order.
4753 lowerGR128Binary(DAG, DL, VT, SystemZISD::SMUL_LOHI,
4754 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4755 else {
4756 // Do a full 128-bit multiplication based on SystemZISD::UMUL_LOHI:
4757 //
4758 // (ll * rl) + ((lh * rl) << 64) + ((ll * rh) << 64)
4759 //
4760 // but using the fact that the upper halves are either all zeros
4761 // or all ones:
4762 //
4763 // (ll * rl) - ((lh & rl) << 64) - ((ll & rh) << 64)
4764 //
4765 // and grouping the right terms together since they are quicker than the
4766 // multiplication:
4767 //
4768 // (ll * rl) - (((lh & rl) + (ll & rh)) << 64)
4769 SDValue C63 = DAG.getConstant(63, DL, MVT::i64);
4770 SDValue LL = Op.getOperand(0);
4771 SDValue RL = Op.getOperand(1);
4772 SDValue LH = DAG.getNode(ISD::SRA, DL, VT, LL, C63);
4773 SDValue RH = DAG.getNode(ISD::SRA, DL, VT, RL, C63);
4774 // SystemZISD::UMUL_LOHI returns the low result in the odd register and
4775 // the high result in the even register. ISD::SMUL_LOHI is defined to
4776 // return the low half first, so the results are in reverse order.
4777 lowerGR128Binary(DAG, DL, VT, SystemZISD::UMUL_LOHI,
4778 LL, RL, Ops[1], Ops[0]);
4779 SDValue NegLLTimesRH = DAG.getNode(ISD::AND, DL, VT, LL, RH);
4780 SDValue NegLHTimesRL = DAG.getNode(ISD::AND, DL, VT, LH, RL);
4781 SDValue NegSum = DAG.getNode(ISD::ADD, DL, VT, NegLLTimesRH, NegLHTimesRL);
4782 Ops[1] = DAG.getNode(ISD::SUB, DL, VT, Ops[1], NegSum);
4783 }
4784 return DAG.getMergeValues(Ops, DL);
4785}
4786
4787SDValue SystemZTargetLowering::lowerUMUL_LOHI(SDValue Op,
4788 SelectionDAG &DAG) const {
4789 EVT VT = Op.getValueType();
4790 SDLoc DL(Op);
4791 SDValue Ops[2];
4792 if (is32Bit(VT))
4793 // Just do a normal 64-bit multiplication and extract the results.
4794 // We define this so that it can be used for constant division.
4795 lowerMUL_LOHI32(DAG, DL, ISD::ZERO_EXTEND, Op.getOperand(0),
4796 Op.getOperand(1), Ops[1], Ops[0]);
4797 else
4798 // SystemZISD::UMUL_LOHI returns the low result in the odd register and
4799 // the high result in the even register. ISD::UMUL_LOHI is defined to
4800 // return the low half first, so the results are in reverse order.
4801 lowerGR128Binary(DAG, DL, VT, SystemZISD::UMUL_LOHI,
4802 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4803 return DAG.getMergeValues(Ops, DL);
4804}
4805
4806SDValue SystemZTargetLowering::lowerSDIVREM(SDValue Op,
4807 SelectionDAG &DAG) const {
4808 SDValue Op0 = Op.getOperand(0);
4809 SDValue Op1 = Op.getOperand(1);
4810 EVT VT = Op.getValueType();
4811 SDLoc DL(Op);
4812
4813 // We use DSGF for 32-bit division. This means the first operand must
4814 // always be 64-bit, and the second operand should be 32-bit whenever
4815 // that is possible, to improve performance.
4816 if (is32Bit(VT))
4817 Op0 = DAG.getNode(ISD::SIGN_EXTEND, DL, MVT::i64, Op0);
4818 else if (DAG.ComputeNumSignBits(Op1) > 32)
4819 Op1 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Op1);
4820
4821 // DSG(F) returns the remainder in the even register and the
4822 // quotient in the odd register.
4823 SDValue Ops[2];
4824 lowerGR128Binary(DAG, DL, VT, SystemZISD::SDIVREM, Op0, Op1, Ops[1], Ops[0]);
4825 return DAG.getMergeValues(Ops, DL);
4826}
4827
4828SDValue SystemZTargetLowering::lowerUDIVREM(SDValue Op,
4829 SelectionDAG &DAG) const {
4830 EVT VT = Op.getValueType();
4831 SDLoc DL(Op);
4832
4833 // DL(G) returns the remainder in the even register and the
4834 // quotient in the odd register.
4835 SDValue Ops[2];
4836 lowerGR128Binary(DAG, DL, VT, SystemZISD::UDIVREM,
4837 Op.getOperand(0), Op.getOperand(1), Ops[1], Ops[0]);
4838 return DAG.getMergeValues(Ops, DL);
4839}
4840
4841SDValue SystemZTargetLowering::lowerOR(SDValue Op, SelectionDAG &DAG) const {
4842 assert(Op.getValueType() == MVT::i64 && "Should be 64-bit operation");
4843
4844 // Get the known-zero masks for each operand.
4845 SDValue Ops[] = {Op.getOperand(0), Op.getOperand(1)};
4846 KnownBits Known[2] = {DAG.computeKnownBits(Ops[0]),
4847 DAG.computeKnownBits(Ops[1])};
4848
4849 // See if the upper 32 bits of one operand and the lower 32 bits of the
4850 // other are known zero. They are the low and high operands respectively.
4851 uint64_t Masks[] = { Known[0].Zero.getZExtValue(),
4852 Known[1].Zero.getZExtValue() };
4853 unsigned High, Low;
4854 if ((Masks[0] >> 32) == 0xffffffff && uint32_t(Masks[1]) == 0xffffffff)
4855 High = 1, Low = 0;
4856 else if ((Masks[1] >> 32) == 0xffffffff && uint32_t(Masks[0]) == 0xffffffff)
4857 High = 0, Low = 1;
4858 else
4859 return Op;
4860
4861 SDValue LowOp = Ops[Low];
4862 SDValue HighOp = Ops[High];
4863
4864 // If the high part is a constant, we're better off using IILH.
4865 if (HighOp.getOpcode() == ISD::Constant)
4866 return Op;
4867
4868 // If the low part is a constant that is outside the range of LHI,
4869 // then we're better off using IILF.
4870 if (LowOp.getOpcode() == ISD::Constant) {
4871 int64_t Value = int32_t(LowOp->getAsZExtVal());
4872 if (!isInt<16>(Value))
4873 return Op;
4874 }
4875
4876 // Check whether the high part is an AND that doesn't change the
4877 // high 32 bits and just masks out low bits. We can skip it if so.
4878 if (HighOp.getOpcode() == ISD::AND &&
4879 HighOp.getOperand(1).getOpcode() == ISD::Constant) {
4880 SDValue HighOp0 = HighOp.getOperand(0);
4882 if (DAG.MaskedValueIsZero(HighOp0, APInt(64, ~(Mask | 0xffffffff))))
4883 HighOp = HighOp0;
4884 }
4885
4886 // Take advantage of the fact that all GR32 operations only change the
4887 // low 32 bits by truncating Low to an i32 and inserting it directly
4888 // using a subreg. The interesting cases are those where the truncation
4889 // can be folded.
4890 SDLoc DL(Op);
4891 SDValue Low32 = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, LowOp);
4892 return DAG.getTargetInsertSubreg(SystemZ::subreg_l32, DL,
4893 MVT::i64, HighOp, Low32);
4894}
4895
4896// Lower SADDO/SSUBO/UADDO/USUBO nodes.
4897SDValue SystemZTargetLowering::lowerXALUO(SDValue Op,
4898 SelectionDAG &DAG) const {
4899 SDNode *N = Op.getNode();
4900 SDValue LHS = N->getOperand(0);
4901 SDValue RHS = N->getOperand(1);
4902 SDLoc DL(N);
4903
4904 if (N->getValueType(0) == MVT::i128) {
4905 unsigned BaseOp = 0;
4906 unsigned FlagOp = 0;
4907 bool IsBorrow = false;
4908 switch (Op.getOpcode()) {
4909 default: llvm_unreachable("Unknown instruction!");
4910 case ISD::UADDO:
4911 BaseOp = ISD::ADD;
4912 FlagOp = SystemZISD::VACC;
4913 break;
4914 case ISD::USUBO:
4915 BaseOp = ISD::SUB;
4916 FlagOp = SystemZISD::VSCBI;
4917 IsBorrow = true;
4918 break;
4919 }
4920 SDValue Result = DAG.getNode(BaseOp, DL, MVT::i128, LHS, RHS);
4921 SDValue Flag = DAG.getNode(FlagOp, DL, MVT::i128, LHS, RHS);
4922 Flag = DAG.getNode(ISD::AssertZext, DL, MVT::i128, Flag,
4923 DAG.getValueType(MVT::i1));
4924 Flag = DAG.getZExtOrTrunc(Flag, DL, N->getValueType(1));
4925 if (IsBorrow)
4926 Flag = DAG.getNode(ISD::XOR, DL, Flag.getValueType(),
4927 Flag, DAG.getConstant(1, DL, Flag.getValueType()));
4928 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, Flag);
4929 }
4930
4931 unsigned BaseOp = 0;
4932 unsigned CCValid = 0;
4933 unsigned CCMask = 0;
4934
4935 switch (Op.getOpcode()) {
4936 default: llvm_unreachable("Unknown instruction!");
4937 case ISD::SADDO:
4938 BaseOp = SystemZISD::SADDO;
4939 CCValid = SystemZ::CCMASK_ARITH;
4941 break;
4942 case ISD::SSUBO:
4943 BaseOp = SystemZISD::SSUBO;
4944 CCValid = SystemZ::CCMASK_ARITH;
4946 break;
4947 case ISD::UADDO:
4948 BaseOp = SystemZISD::UADDO;
4949 CCValid = SystemZ::CCMASK_LOGICAL;
4951 break;
4952 case ISD::USUBO:
4953 BaseOp = SystemZISD::USUBO;
4954 CCValid = SystemZ::CCMASK_LOGICAL;
4956 break;
4957 }
4958
4959 SDVTList VTs = DAG.getVTList(N->getValueType(0), MVT::i32);
4960 SDValue Result = DAG.getNode(BaseOp, DL, VTs, LHS, RHS);
4961
4962 SDValue SetCC = emitSETCC(DAG, DL, Result.getValue(1), CCValid, CCMask);
4963 if (N->getValueType(1) == MVT::i1)
4964 SetCC = DAG.getNode(ISD::TRUNCATE, DL, MVT::i1, SetCC);
4965
4966 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, SetCC);
4967}
4968
4969static bool isAddCarryChain(SDValue Carry) {
4970 while (Carry.getOpcode() == ISD::UADDO_CARRY &&
4971 Carry->getValueType(0) != MVT::i128)
4972 Carry = Carry.getOperand(2);
4973 return Carry.getOpcode() == ISD::UADDO &&
4974 Carry->getValueType(0) != MVT::i128;
4975}
4976
4977static bool isSubBorrowChain(SDValue Carry) {
4978 while (Carry.getOpcode() == ISD::USUBO_CARRY &&
4979 Carry->getValueType(0) != MVT::i128)
4980 Carry = Carry.getOperand(2);
4981 return Carry.getOpcode() == ISD::USUBO &&
4982 Carry->getValueType(0) != MVT::i128;
4983}
4984
4985// Lower UADDO_CARRY/USUBO_CARRY nodes.
4986SDValue SystemZTargetLowering::lowerUADDSUBO_CARRY(SDValue Op,
4987 SelectionDAG &DAG) const {
4988
4989 SDNode *N = Op.getNode();
4990 MVT VT = N->getSimpleValueType(0);
4991
4992 // Let legalize expand this if it isn't a legal type yet.
4993 if (!DAG.getTargetLoweringInfo().isTypeLegal(VT))
4994 return SDValue();
4995
4996 SDValue LHS = N->getOperand(0);
4997 SDValue RHS = N->getOperand(1);
4998 SDValue Carry = Op.getOperand(2);
4999 SDLoc DL(N);
5000
5001 if (VT == MVT::i128) {
5002 unsigned BaseOp = 0;
5003 unsigned FlagOp = 0;
5004 bool IsBorrow = false;
5005 switch (Op.getOpcode()) {
5006 default: llvm_unreachable("Unknown instruction!");
5007 case ISD::UADDO_CARRY:
5008 BaseOp = SystemZISD::VAC;
5009 FlagOp = SystemZISD::VACCC;
5010 break;
5011 case ISD::USUBO_CARRY:
5012 BaseOp = SystemZISD::VSBI;
5013 FlagOp = SystemZISD::VSBCBI;
5014 IsBorrow = true;
5015 break;
5016 }
5017 if (IsBorrow)
5018 Carry = DAG.getNode(ISD::XOR, DL, Carry.getValueType(),
5019 Carry, DAG.getConstant(1, DL, Carry.getValueType()));
5020 Carry = DAG.getZExtOrTrunc(Carry, DL, MVT::i128);
5021 SDValue Result = DAG.getNode(BaseOp, DL, MVT::i128, LHS, RHS, Carry);
5022 SDValue Flag = DAG.getNode(FlagOp, DL, MVT::i128, LHS, RHS, Carry);
5023 Flag = DAG.getNode(ISD::AssertZext, DL, MVT::i128, Flag,
5024 DAG.getValueType(MVT::i1));
5025 Flag = DAG.getZExtOrTrunc(Flag, DL, N->getValueType(1));
5026 if (IsBorrow)
5027 Flag = DAG.getNode(ISD::XOR, DL, Flag.getValueType(),
5028 Flag, DAG.getConstant(1, DL, Flag.getValueType()));
5029 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, Flag);
5030 }
5031
5032 unsigned BaseOp = 0;
5033 unsigned CCValid = 0;
5034 unsigned CCMask = 0;
5035
5036 switch (Op.getOpcode()) {
5037 default: llvm_unreachable("Unknown instruction!");
5038 case ISD::UADDO_CARRY:
5039 if (!isAddCarryChain(Carry))
5040 return SDValue();
5041
5042 BaseOp = SystemZISD::ADDCARRY;
5043 CCValid = SystemZ::CCMASK_LOGICAL;
5045 break;
5046 case ISD::USUBO_CARRY:
5047 if (!isSubBorrowChain(Carry))
5048 return SDValue();
5049
5050 BaseOp = SystemZISD::SUBCARRY;
5051 CCValid = SystemZ::CCMASK_LOGICAL;
5053 break;
5054 }
5055
5056 // Set the condition code from the carry flag.
5057 Carry = DAG.getNode(SystemZISD::GET_CCMASK, DL, MVT::i32, Carry,
5058 DAG.getConstant(CCValid, DL, MVT::i32),
5059 DAG.getConstant(CCMask, DL, MVT::i32));
5060
5061 SDVTList VTs = DAG.getVTList(VT, MVT::i32);
5062 SDValue Result = DAG.getNode(BaseOp, DL, VTs, LHS, RHS, Carry);
5063
5064 SDValue SetCC = emitSETCC(DAG, DL, Result.getValue(1), CCValid, CCMask);
5065 if (N->getValueType(1) == MVT::i1)
5066 SetCC = DAG.getNode(ISD::TRUNCATE, DL, MVT::i1, SetCC);
5067
5068 return DAG.getNode(ISD::MERGE_VALUES, DL, N->getVTList(), Result, SetCC);
5069}
5070
5071SDValue SystemZTargetLowering::lowerCTPOP(SDValue Op,
5072 SelectionDAG &DAG) const {
5073 EVT VT = Op.getValueType();
5074 SDLoc DL(Op);
5075 Op = Op.getOperand(0);
5076
5077 if (VT.getScalarSizeInBits() == 128) {
5078 Op = DAG.getNode(ISD::BITCAST, DL, MVT::v2i64, Op);
5079 Op = DAG.getNode(ISD::CTPOP, DL, MVT::v2i64, Op);
5080 SDValue Tmp = DAG.getSplatBuildVector(MVT::v2i64, DL,
5081 DAG.getConstant(0, DL, MVT::i64));
5082 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5083 return Op;
5084 }
5085
5086 // Handle vector types via VPOPCT.
5087 if (VT.isVector()) {
5088 Op = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, Op);
5089 Op = DAG.getNode(SystemZISD::POPCNT, DL, MVT::v16i8, Op);
5090 switch (VT.getScalarSizeInBits()) {
5091 case 8:
5092 break;
5093 case 16: {
5094 Op = DAG.getNode(ISD::BITCAST, DL, VT, Op);
5095 SDValue Shift = DAG.getConstant(8, DL, MVT::i32);
5096 SDValue Tmp = DAG.getNode(SystemZISD::VSHL_BY_SCALAR, DL, VT, Op, Shift);
5097 Op = DAG.getNode(ISD::ADD, DL, VT, Op, Tmp);
5098 Op = DAG.getNode(SystemZISD::VSRL_BY_SCALAR, DL, VT, Op, Shift);
5099 break;
5100 }
5101 case 32: {
5102 SDValue Tmp = DAG.getSplatBuildVector(MVT::v16i8, DL,
5103 DAG.getConstant(0, DL, MVT::i32));
5104 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5105 break;
5106 }
5107 case 64: {
5108 SDValue Tmp = DAG.getSplatBuildVector(MVT::v16i8, DL,
5109 DAG.getConstant(0, DL, MVT::i32));
5110 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::v4i32, Op, Tmp);
5111 Tmp = DAG.getNode(ISD::BITCAST, DL, MVT::v4i32, Tmp);
5112 Op = DAG.getNode(SystemZISD::VSUM, DL, VT, Op, Tmp);
5113 break;
5114 }
5115 default:
5116 llvm_unreachable("Unexpected type");
5117 }
5118 return Op;
5119 }
5120
5121 // Get the known-zero mask for the operand.
5122 KnownBits Known = DAG.computeKnownBits(Op);
5123 unsigned NumSignificantBits = Known.getMaxValue().getActiveBits();
5124 if (NumSignificantBits == 0)
5125 return DAG.getConstant(0, DL, VT);
5126
5127 // Skip known-zero high parts of the operand.
5128 int64_t OrigBitSize = VT.getSizeInBits();
5129 int64_t BitSize = llvm::bit_ceil(NumSignificantBits);
5130 BitSize = std::min(BitSize, OrigBitSize);
5131
5132 // The POPCNT instruction counts the number of bits in each byte.
5133 Op = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op);
5134 Op = DAG.getNode(SystemZISD::POPCNT, DL, MVT::i64, Op);
5135 Op = DAG.getNode(ISD::TRUNCATE, DL, VT, Op);
5136
5137 // Add up per-byte counts in a binary tree. All bits of Op at
5138 // position larger than BitSize remain zero throughout.
5139 for (int64_t I = BitSize / 2; I >= 8; I = I / 2) {
5140 SDValue Tmp = DAG.getNode(ISD::SHL, DL, VT, Op, DAG.getConstant(I, DL, VT));
5141 if (BitSize != OrigBitSize)
5142 Tmp = DAG.getNode(ISD::AND, DL, VT, Tmp,
5143 DAG.getConstant(((uint64_t)1 << BitSize) - 1, DL, VT));
5144 Op = DAG.getNode(ISD::ADD, DL, VT, Op, Tmp);
5145 }
5146
5147 // Extract overall result from high byte.
5148 if (BitSize > 8)
5149 Op = DAG.getNode(ISD::SRL, DL, VT, Op,
5150 DAG.getConstant(BitSize - 8, DL, VT));
5151
5152 return Op;
5153}
5154
5155SDValue SystemZTargetLowering::lowerATOMIC_FENCE(SDValue Op,
5156 SelectionDAG &DAG) const {
5157 SDLoc DL(Op);
5158 AtomicOrdering FenceOrdering =
5159 static_cast<AtomicOrdering>(Op.getConstantOperandVal(1));
5160 SyncScope::ID FenceSSID =
5161 static_cast<SyncScope::ID>(Op.getConstantOperandVal(2));
5162
5163 // The only fence that needs an instruction is a sequentially-consistent
5164 // cross-thread fence.
5165 if (FenceOrdering == AtomicOrdering::SequentiallyConsistent &&
5166 FenceSSID == SyncScope::System) {
5167 return SDValue(DAG.getMachineNode(SystemZ::Serialize, DL, MVT::Other,
5168 Op.getOperand(0)),
5169 0);
5170 }
5171
5172 // MEMBARRIER is a compiler barrier; it codegens to a no-op.
5173 return DAG.getNode(ISD::MEMBARRIER, DL, MVT::Other, Op.getOperand(0));
5174}
5175
5176SDValue SystemZTargetLowering::lowerATOMIC_LOAD(SDValue Op,
5177 SelectionDAG &DAG) const {
5178 EVT RegVT = Op.getValueType();
5179 if (RegVT.getSizeInBits() == 128)
5180 return lowerATOMIC_LDST_I128(Op, DAG);
5181 return lowerLoadF16(Op, DAG);
5182}
5183
5184SDValue SystemZTargetLowering::lowerATOMIC_STORE(SDValue Op,
5185 SelectionDAG &DAG) const {
5186 auto *Node = cast<AtomicSDNode>(Op.getNode());
5187 if (Node->getMemoryVT().getSizeInBits() == 128)
5188 return lowerATOMIC_LDST_I128(Op, DAG);
5189 return lowerStoreF16(Op, DAG);
5190}
5191
5192SDValue SystemZTargetLowering::lowerATOMIC_LDST_I128(SDValue Op,
5193 SelectionDAG &DAG) const {
5194 auto *Node = cast<AtomicSDNode>(Op.getNode());
5195 assert(
5196 (Node->getMemoryVT() == MVT::i128 || Node->getMemoryVT() == MVT::f128) &&
5197 "Only custom lowering i128 or f128.");
5198 // Use same code to handle both legal and non-legal i128 types.
5200 LowerOperationWrapper(Node, Results, DAG);
5201 return DAG.getMergeValues(Results, SDLoc(Op));
5202}
5203
5204// Prepare for a Compare And Swap for a subword operation. This needs to be
5205// done in memory with 4 bytes at natural alignment.
5207 SDValue &AlignedAddr, SDValue &BitShift,
5208 SDValue &NegBitShift) {
5209 EVT PtrVT = Addr.getValueType();
5210 EVT WideVT = MVT::i32;
5211
5212 // Get the address of the containing word.
5213 AlignedAddr = DAG.getNode(ISD::AND, DL, PtrVT, Addr,
5214 DAG.getSignedConstant(-4, DL, PtrVT));
5215
5216 // Get the number of bits that the word must be rotated left in order
5217 // to bring the field to the top bits of a GR32.
5218 BitShift = DAG.getNode(ISD::SHL, DL, PtrVT, Addr,
5219 DAG.getConstant(3, DL, PtrVT));
5220 BitShift = DAG.getNode(ISD::TRUNCATE, DL, WideVT, BitShift);
5221
5222 // Get the complementing shift amount, for rotating a field in the top
5223 // bits back to its proper position.
5224 NegBitShift = DAG.getNode(ISD::SUB, DL, WideVT,
5225 DAG.getConstant(0, DL, WideVT), BitShift);
5226
5227}
5228
5229// Op is an 8-, 16-bit or 32-bit ATOMIC_LOAD_* operation. Lower the first
5230// two into the fullword ATOMIC_LOADW_* operation given by Opcode.
5231SDValue SystemZTargetLowering::lowerATOMIC_LOAD_OP(SDValue Op,
5232 SelectionDAG &DAG,
5233 unsigned Opcode) const {
5234 auto *Node = cast<AtomicSDNode>(Op.getNode());
5235
5236 // 32-bit operations need no special handling.
5237 EVT NarrowVT = Node->getMemoryVT();
5238 EVT WideVT = MVT::i32;
5239 if (NarrowVT == WideVT)
5240 return Op;
5241
5242 int64_t BitSize = NarrowVT.getSizeInBits();
5243 SDValue ChainIn = Node->getChain();
5244 SDValue Addr = Node->getBasePtr();
5245 SDValue Src2 = Node->getVal();
5246 MachineMemOperand *MMO = Node->getMemOperand();
5247 SDLoc DL(Node);
5248
5249 // Convert atomic subtracts of constants into additions.
5250 if (Opcode == SystemZISD::ATOMIC_LOADW_SUB)
5251 if (auto *Const = dyn_cast<ConstantSDNode>(Src2)) {
5252 Opcode = SystemZISD::ATOMIC_LOADW_ADD;
5253 Src2 = DAG.getSignedConstant(-Const->getSExtValue(), DL,
5254 Src2.getValueType());
5255 }
5256
5257 SDValue AlignedAddr, BitShift, NegBitShift;
5258 getCSAddressAndShifts(Addr, DAG, DL, AlignedAddr, BitShift, NegBitShift);
5259
5260 // Extend the source operand to 32 bits and prepare it for the inner loop.
5261 // ATOMIC_SWAPW uses RISBG to rotate the field left, but all other
5262 // operations require the source to be shifted in advance. (This shift
5263 // can be folded if the source is constant.) For AND and NAND, the lower
5264 // bits must be set, while for other opcodes they should be left clear.
5265 if (Opcode != SystemZISD::ATOMIC_SWAPW)
5266 Src2 = DAG.getNode(ISD::SHL, DL, WideVT, Src2,
5267 DAG.getConstant(32 - BitSize, DL, WideVT));
5268 if (Opcode == SystemZISD::ATOMIC_LOADW_AND ||
5269 Opcode == SystemZISD::ATOMIC_LOADW_NAND)
5270 Src2 = DAG.getNode(ISD::OR, DL, WideVT, Src2,
5271 DAG.getConstant(uint32_t(-1) >> BitSize, DL, WideVT));
5272
5273 // Construct the ATOMIC_LOADW_* node.
5274 SDVTList VTList = DAG.getVTList(WideVT, MVT::Other);
5275 SDValue Ops[] = { ChainIn, AlignedAddr, Src2, BitShift, NegBitShift,
5276 DAG.getConstant(BitSize, DL, WideVT) };
5277 SDValue AtomicOp = DAG.getMemIntrinsicNode(Opcode, DL, VTList, Ops,
5278 NarrowVT, MMO);
5279
5280 // Rotate the result of the final CS so that the field is in the lower
5281 // bits of a GR32, then truncate it.
5282 SDValue ResultShift = DAG.getNode(ISD::ADD, DL, WideVT, BitShift,
5283 DAG.getConstant(BitSize, DL, WideVT));
5284 SDValue Result = DAG.getNode(ISD::ROTL, DL, WideVT, AtomicOp, ResultShift);
5285
5286 SDValue RetOps[2] = { Result, AtomicOp.getValue(1) };
5287 return DAG.getMergeValues(RetOps, DL);
5288}
5289
5290// Op is an ATOMIC_LOAD_SUB operation. Lower 8- and 16-bit operations into
5291// ATOMIC_LOADW_SUBs and convert 32- and 64-bit operations into additions.
5292SDValue SystemZTargetLowering::lowerATOMIC_LOAD_SUB(SDValue Op,
5293 SelectionDAG &DAG) const {
5294 auto *Node = cast<AtomicSDNode>(Op.getNode());
5295 EVT MemVT = Node->getMemoryVT();
5296 if (MemVT == MVT::i32 || MemVT == MVT::i64) {
5297 // A full-width operation: negate and use LAA(G).
5298 assert(Op.getValueType() == MemVT && "Mismatched VTs");
5299 assert(Subtarget.hasInterlockedAccess1() &&
5300 "Should have been expanded by AtomicExpand pass.");
5301 SDValue Src2 = Node->getVal();
5302 SDLoc DL(Src2);
5303 SDValue NegSrc2 =
5304 DAG.getNode(ISD::SUB, DL, MemVT, DAG.getConstant(0, DL, MemVT), Src2);
5305 return DAG.getAtomic(ISD::ATOMIC_LOAD_ADD, DL, MemVT,
5306 Node->getChain(), Node->getBasePtr(), NegSrc2,
5307 Node->getMemOperand());
5308 }
5309
5310 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_SUB);
5311}
5312
5313// Lower 8/16/32/64-bit ATOMIC_CMP_SWAP_WITH_SUCCESS node.
5314SDValue SystemZTargetLowering::lowerATOMIC_CMP_SWAP(SDValue Op,
5315 SelectionDAG &DAG) const {
5316 auto *Node = cast<AtomicSDNode>(Op.getNode());
5317 SDValue ChainIn = Node->getOperand(0);
5318 SDValue Addr = Node->getOperand(1);
5319 SDValue CmpVal = Node->getOperand(2);
5320 SDValue SwapVal = Node->getOperand(3);
5321 MachineMemOperand *MMO = Node->getMemOperand();
5322 SDLoc DL(Node);
5323
5324 if (Node->getMemoryVT() == MVT::i128) {
5325 // Use same code to handle both legal and non-legal i128 types.
5327 LowerOperationWrapper(Node, Results, DAG);
5328 return DAG.getMergeValues(Results, DL);
5329 }
5330
5331 // We have native support for 32-bit and 64-bit compare and swap, but we
5332 // still need to expand extracting the "success" result from the CC.
5333 EVT NarrowVT = Node->getMemoryVT();
5334 EVT WideVT = NarrowVT == MVT::i64 ? MVT::i64 : MVT::i32;
5335 if (NarrowVT == WideVT) {
5336 SDVTList Tys = DAG.getVTList(WideVT, MVT::i32, MVT::Other);
5337 SDValue Ops[] = { ChainIn, Addr, CmpVal, SwapVal };
5338 SDValue AtomicOp = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAP,
5339 DL, Tys, Ops, NarrowVT, MMO);
5340 SDValue Success = emitSETCC(DAG, DL, AtomicOp.getValue(1),
5342
5343 DAG.ReplaceAllUsesOfValueWith(Op.getValue(0), AtomicOp.getValue(0));
5344 DAG.ReplaceAllUsesOfValueWith(Op.getValue(1), Success);
5345 DAG.ReplaceAllUsesOfValueWith(Op.getValue(2), AtomicOp.getValue(2));
5346 return SDValue();
5347 }
5348
5349 // Convert 8-bit and 16-bit compare and swap to a loop, implemented
5350 // via a fullword ATOMIC_CMP_SWAPW operation.
5351 int64_t BitSize = NarrowVT.getSizeInBits();
5352
5353 SDValue AlignedAddr, BitShift, NegBitShift;
5354 getCSAddressAndShifts(Addr, DAG, DL, AlignedAddr, BitShift, NegBitShift);
5355
5356 // Construct the ATOMIC_CMP_SWAPW node.
5357 SDVTList VTList = DAG.getVTList(WideVT, MVT::i32, MVT::Other);
5358 SDValue Ops[] = { ChainIn, AlignedAddr, CmpVal, SwapVal, BitShift,
5359 NegBitShift, DAG.getConstant(BitSize, DL, WideVT) };
5360 SDValue AtomicOp = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAPW, DL,
5361 VTList, Ops, NarrowVT, MMO);
5362 SDValue Success = emitSETCC(DAG, DL, AtomicOp.getValue(1),
5364
5365 // emitAtomicCmpSwapW() will zero extend the result (original value).
5366 SDValue OrigVal = DAG.getNode(ISD::AssertZext, DL, WideVT, AtomicOp.getValue(0),
5367 DAG.getValueType(NarrowVT));
5368 DAG.ReplaceAllUsesOfValueWith(Op.getValue(0), OrigVal);
5369 DAG.ReplaceAllUsesOfValueWith(Op.getValue(1), Success);
5370 DAG.ReplaceAllUsesOfValueWith(Op.getValue(2), AtomicOp.getValue(2));
5371 return SDValue();
5372}
5373
5375SystemZTargetLowering::getTargetMMOFlags(const Instruction &I) const {
5376 // Because of how we convert atomic_load and atomic_store to normal loads and
5377 // stores in the DAG, we need to ensure that the MMOs are marked volatile
5378 // since DAGCombine hasn't been updated to account for atomic, but non
5379 // volatile loads. (See D57601)
5380 if (auto *SI = dyn_cast<StoreInst>(&I))
5381 if (SI->isAtomic())
5383 if (auto *LI = dyn_cast<LoadInst>(&I))
5384 if (LI->isAtomic())
5386 if (auto *AI = dyn_cast<AtomicRMWInst>(&I))
5387 if (AI->isAtomic())
5389 if (auto *AI = dyn_cast<AtomicCmpXchgInst>(&I))
5390 if (AI->isAtomic())
5393}
5394
5395SDValue SystemZTargetLowering::lowerSTACKSAVE(SDValue Op,
5396 SelectionDAG &DAG) const {
5398 auto *Regs = Subtarget.getSpecialRegisters();
5400 report_fatal_error("Variable-sized stack allocations are not supported "
5401 "in GHC calling convention");
5402 return DAG.getCopyFromReg(Op.getOperand(0), SDLoc(Op),
5403 Regs->getStackPointerRegister(), Op.getValueType());
5404}
5405
5406SDValue SystemZTargetLowering::lowerSTACKRESTORE(SDValue Op,
5407 SelectionDAG &DAG) const {
5409 auto *Regs = Subtarget.getSpecialRegisters();
5410 bool StoreBackchain = MF.getSubtarget<SystemZSubtarget>().hasBackChain();
5411
5413 report_fatal_error("Variable-sized stack allocations are not supported "
5414 "in GHC calling convention");
5415
5416 SDValue Chain = Op.getOperand(0);
5417 SDValue NewSP = Op.getOperand(1);
5418 SDValue Backchain;
5419 SDLoc DL(Op);
5420
5421 if (StoreBackchain) {
5422 SDValue OldSP = DAG.getCopyFromReg(
5423 Chain, DL, Regs->getStackPointerRegister(), MVT::i64);
5424 Backchain = DAG.getLoad(MVT::i64, DL, Chain, getBackchainAddress(OldSP, DAG),
5425 MachinePointerInfo());
5426 }
5427
5428 Chain = DAG.getCopyToReg(Chain, DL, Regs->getStackPointerRegister(), NewSP);
5429
5430 if (StoreBackchain)
5431 Chain = DAG.getStore(Chain, DL, Backchain, getBackchainAddress(NewSP, DAG),
5432 MachinePointerInfo());
5433
5434 return Chain;
5435}
5436
5437SDValue SystemZTargetLowering::lowerPREFETCH(SDValue Op,
5438 SelectionDAG &DAG) const {
5439 bool IsData = Op.getConstantOperandVal(4);
5440 if (!IsData)
5441 // Just preserve the chain.
5442 return Op.getOperand(0);
5443
5444 SDLoc DL(Op);
5445 bool IsWrite = Op.getConstantOperandVal(2);
5446 unsigned Code = IsWrite ? SystemZ::PFD_WRITE : SystemZ::PFD_READ;
5447 auto *Node = cast<MemIntrinsicSDNode>(Op.getNode());
5448 SDValue Ops[] = {Op.getOperand(0), DAG.getTargetConstant(Code, DL, MVT::i32),
5449 Op.getOperand(1)};
5450 return DAG.getMemIntrinsicNode(SystemZISD::PREFETCH, DL,
5451 Node->getVTList(), Ops,
5452 Node->getMemoryVT(), Node->getMemOperand());
5453}
5454
5455SDValue
5456SystemZTargetLowering::lowerINTRINSIC_W_CHAIN(SDValue Op,
5457 SelectionDAG &DAG) const {
5458 unsigned Opcode, CCValid;
5459 if (isIntrinsicWithCCAndChain(Op, Opcode, CCValid)) {
5460 assert(Op->getNumValues() == 2 && "Expected only CC result and chain");
5461 SDNode *Node = emitIntrinsicWithCCAndChain(DAG, Op, Opcode);
5462 SDValue CC = getCCResult(DAG, SDValue(Node, 0));
5463 DAG.ReplaceAllUsesOfValueWith(SDValue(Op.getNode(), 0), CC);
5464 return SDValue();
5465 }
5466
5467 return SDValue();
5468}
5469
5470SDValue
5471SystemZTargetLowering::lowerINTRINSIC_WO_CHAIN(SDValue Op,
5472 SelectionDAG &DAG) const {
5473 unsigned Opcode, CCValid;
5474 if (isIntrinsicWithCC(Op, Opcode, CCValid)) {
5475 SDNode *Node = emitIntrinsicWithCC(DAG, Op, Opcode);
5476 if (Op->getNumValues() == 1)
5477 return getCCResult(DAG, SDValue(Node, 0));
5478 assert(Op->getNumValues() == 2 && "Expected a CC and non-CC result");
5479 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op), Op->getVTList(),
5480 SDValue(Node, 0), getCCResult(DAG, SDValue(Node, 1)));
5481 }
5482
5483 unsigned Id = Op.getConstantOperandVal(0);
5484 switch (Id) {
5485 case Intrinsic::thread_pointer:
5486 return lowerThreadPointer(SDLoc(Op), DAG);
5487
5488 case Intrinsic::s390_vpdi:
5489 return DAG.getNode(SystemZISD::PERMUTE_DWORDS, SDLoc(Op), Op.getValueType(),
5490 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5491
5492 case Intrinsic::s390_vperm:
5493 return DAG.getNode(SystemZISD::PERMUTE, SDLoc(Op), Op.getValueType(),
5494 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5495
5496 case Intrinsic::s390_vuphb:
5497 case Intrinsic::s390_vuphh:
5498 case Intrinsic::s390_vuphf:
5499 case Intrinsic::s390_vuphg:
5500 return DAG.getNode(SystemZISD::UNPACK_HIGH, SDLoc(Op), Op.getValueType(),
5501 Op.getOperand(1));
5502
5503 case Intrinsic::s390_vuplhb:
5504 case Intrinsic::s390_vuplhh:
5505 case Intrinsic::s390_vuplhf:
5506 case Intrinsic::s390_vuplhg:
5507 return DAG.getNode(SystemZISD::UNPACKL_HIGH, SDLoc(Op), Op.getValueType(),
5508 Op.getOperand(1));
5509
5510 case Intrinsic::s390_vuplb:
5511 case Intrinsic::s390_vuplhw:
5512 case Intrinsic::s390_vuplf:
5513 case Intrinsic::s390_vuplg:
5514 return DAG.getNode(SystemZISD::UNPACK_LOW, SDLoc(Op), Op.getValueType(),
5515 Op.getOperand(1));
5516
5517 case Intrinsic::s390_vupllb:
5518 case Intrinsic::s390_vupllh:
5519 case Intrinsic::s390_vupllf:
5520 case Intrinsic::s390_vupllg:
5521 return DAG.getNode(SystemZISD::UNPACKL_LOW, SDLoc(Op), Op.getValueType(),
5522 Op.getOperand(1));
5523
5524 case Intrinsic::s390_vsumb:
5525 case Intrinsic::s390_vsumh:
5526 case Intrinsic::s390_vsumgh:
5527 case Intrinsic::s390_vsumgf:
5528 case Intrinsic::s390_vsumqf:
5529 case Intrinsic::s390_vsumqg:
5530 return DAG.getNode(SystemZISD::VSUM, SDLoc(Op), Op.getValueType(),
5531 Op.getOperand(1), Op.getOperand(2));
5532
5533 case Intrinsic::s390_vaq:
5534 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5535 Op.getOperand(1), Op.getOperand(2));
5536 case Intrinsic::s390_vaccb:
5537 case Intrinsic::s390_vacch:
5538 case Intrinsic::s390_vaccf:
5539 case Intrinsic::s390_vaccg:
5540 case Intrinsic::s390_vaccq:
5541 return DAG.getNode(SystemZISD::VACC, SDLoc(Op), Op.getValueType(),
5542 Op.getOperand(1), Op.getOperand(2));
5543 case Intrinsic::s390_vacq:
5544 return DAG.getNode(SystemZISD::VAC, SDLoc(Op), Op.getValueType(),
5545 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5546 case Intrinsic::s390_vacccq:
5547 return DAG.getNode(SystemZISD::VACCC, SDLoc(Op), Op.getValueType(),
5548 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5549
5550 case Intrinsic::s390_vsq:
5551 return DAG.getNode(ISD::SUB, SDLoc(Op), Op.getValueType(),
5552 Op.getOperand(1), Op.getOperand(2));
5553 case Intrinsic::s390_vscbib:
5554 case Intrinsic::s390_vscbih:
5555 case Intrinsic::s390_vscbif:
5556 case Intrinsic::s390_vscbig:
5557 case Intrinsic::s390_vscbiq:
5558 return DAG.getNode(SystemZISD::VSCBI, SDLoc(Op), Op.getValueType(),
5559 Op.getOperand(1), Op.getOperand(2));
5560 case Intrinsic::s390_vsbiq:
5561 return DAG.getNode(SystemZISD::VSBI, SDLoc(Op), Op.getValueType(),
5562 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5563 case Intrinsic::s390_vsbcbiq:
5564 return DAG.getNode(SystemZISD::VSBCBI, SDLoc(Op), Op.getValueType(),
5565 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5566
5567 case Intrinsic::s390_vmhb:
5568 case Intrinsic::s390_vmhh:
5569 case Intrinsic::s390_vmhf:
5570 case Intrinsic::s390_vmhg:
5571 case Intrinsic::s390_vmhq:
5572 return DAG.getNode(ISD::MULHS, SDLoc(Op), Op.getValueType(),
5573 Op.getOperand(1), Op.getOperand(2));
5574 case Intrinsic::s390_vmlhb:
5575 case Intrinsic::s390_vmlhh:
5576 case Intrinsic::s390_vmlhf:
5577 case Intrinsic::s390_vmlhg:
5578 case Intrinsic::s390_vmlhq:
5579 return DAG.getNode(ISD::MULHU, SDLoc(Op), Op.getValueType(),
5580 Op.getOperand(1), Op.getOperand(2));
5581
5582 case Intrinsic::s390_vmahb:
5583 case Intrinsic::s390_vmahh:
5584 case Intrinsic::s390_vmahf:
5585 case Intrinsic::s390_vmahg:
5586 case Intrinsic::s390_vmahq:
5587 return DAG.getNode(SystemZISD::VMAH, SDLoc(Op), Op.getValueType(),
5588 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5589 case Intrinsic::s390_vmalhb:
5590 case Intrinsic::s390_vmalhh:
5591 case Intrinsic::s390_vmalhf:
5592 case Intrinsic::s390_vmalhg:
5593 case Intrinsic::s390_vmalhq:
5594 return DAG.getNode(SystemZISD::VMALH, SDLoc(Op), Op.getValueType(),
5595 Op.getOperand(1), Op.getOperand(2), Op.getOperand(3));
5596
5597 case Intrinsic::s390_vmeb:
5598 case Intrinsic::s390_vmeh:
5599 case Intrinsic::s390_vmef:
5600 case Intrinsic::s390_vmeg:
5601 return DAG.getNode(SystemZISD::VME, SDLoc(Op), Op.getValueType(),
5602 Op.getOperand(1), Op.getOperand(2));
5603 case Intrinsic::s390_vmleb:
5604 case Intrinsic::s390_vmleh:
5605 case Intrinsic::s390_vmlef:
5606 case Intrinsic::s390_vmleg:
5607 return DAG.getNode(SystemZISD::VMLE, SDLoc(Op), Op.getValueType(),
5608 Op.getOperand(1), Op.getOperand(2));
5609 case Intrinsic::s390_vmob:
5610 case Intrinsic::s390_vmoh:
5611 case Intrinsic::s390_vmof:
5612 case Intrinsic::s390_vmog:
5613 return DAG.getNode(SystemZISD::VMO, SDLoc(Op), Op.getValueType(),
5614 Op.getOperand(1), Op.getOperand(2));
5615 case Intrinsic::s390_vmlob:
5616 case Intrinsic::s390_vmloh:
5617 case Intrinsic::s390_vmlof:
5618 case Intrinsic::s390_vmlog:
5619 return DAG.getNode(SystemZISD::VMLO, SDLoc(Op), Op.getValueType(),
5620 Op.getOperand(1), Op.getOperand(2));
5621
5622 case Intrinsic::s390_vmaeb:
5623 case Intrinsic::s390_vmaeh:
5624 case Intrinsic::s390_vmaef:
5625 case Intrinsic::s390_vmaeg:
5626 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5627 DAG.getNode(SystemZISD::VME, SDLoc(Op), Op.getValueType(),
5628 Op.getOperand(1), Op.getOperand(2)),
5629 Op.getOperand(3));
5630 case Intrinsic::s390_vmaleb:
5631 case Intrinsic::s390_vmaleh:
5632 case Intrinsic::s390_vmalef:
5633 case Intrinsic::s390_vmaleg:
5634 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5635 DAG.getNode(SystemZISD::VMLE, SDLoc(Op), Op.getValueType(),
5636 Op.getOperand(1), Op.getOperand(2)),
5637 Op.getOperand(3));
5638 case Intrinsic::s390_vmaob:
5639 case Intrinsic::s390_vmaoh:
5640 case Intrinsic::s390_vmaof:
5641 case Intrinsic::s390_vmaog:
5642 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5643 DAG.getNode(SystemZISD::VMO, SDLoc(Op), Op.getValueType(),
5644 Op.getOperand(1), Op.getOperand(2)),
5645 Op.getOperand(3));
5646 case Intrinsic::s390_vmalob:
5647 case Intrinsic::s390_vmaloh:
5648 case Intrinsic::s390_vmalof:
5649 case Intrinsic::s390_vmalog:
5650 return DAG.getNode(ISD::ADD, SDLoc(Op), Op.getValueType(),
5651 DAG.getNode(SystemZISD::VMLO, SDLoc(Op), Op.getValueType(),
5652 Op.getOperand(1), Op.getOperand(2)),
5653 Op.getOperand(3));
5654 }
5655
5656 return SDValue();
5657}
5658
5659namespace {
5660// Says that SystemZISD operation Opcode can be used to perform the equivalent
5661// of a VPERM with permute vector Bytes. If Opcode takes three operands,
5662// Operand is the constant third operand, otherwise it is the number of
5663// bytes in each element of the result.
5664struct Permute {
5665 unsigned Opcode;
5666 unsigned Operand;
5667 unsigned char Bytes[SystemZ::VectorBytes];
5668};
5669}
5670
5671static const Permute PermuteForms[] = {
5672 // VMRHG
5673 { SystemZISD::MERGE_HIGH, 8,
5674 { 0, 1, 2, 3, 4, 5, 6, 7, 16, 17, 18, 19, 20, 21, 22, 23 } },
5675 // VMRHF
5676 { SystemZISD::MERGE_HIGH, 4,
5677 { 0, 1, 2, 3, 16, 17, 18, 19, 4, 5, 6, 7, 20, 21, 22, 23 } },
5678 // VMRHH
5679 { SystemZISD::MERGE_HIGH, 2,
5680 { 0, 1, 16, 17, 2, 3, 18, 19, 4, 5, 20, 21, 6, 7, 22, 23 } },
5681 // VMRHB
5682 { SystemZISD::MERGE_HIGH, 1,
5683 { 0, 16, 1, 17, 2, 18, 3, 19, 4, 20, 5, 21, 6, 22, 7, 23 } },
5684 // VMRLG
5685 { SystemZISD::MERGE_LOW, 8,
5686 { 8, 9, 10, 11, 12, 13, 14, 15, 24, 25, 26, 27, 28, 29, 30, 31 } },
5687 // VMRLF
5688 { SystemZISD::MERGE_LOW, 4,
5689 { 8, 9, 10, 11, 24, 25, 26, 27, 12, 13, 14, 15, 28, 29, 30, 31 } },
5690 // VMRLH
5691 { SystemZISD::MERGE_LOW, 2,
5692 { 8, 9, 24, 25, 10, 11, 26, 27, 12, 13, 28, 29, 14, 15, 30, 31 } },
5693 // VMRLB
5694 { SystemZISD::MERGE_LOW, 1,
5695 { 8, 24, 9, 25, 10, 26, 11, 27, 12, 28, 13, 29, 14, 30, 15, 31 } },
5696 // VPKG
5697 { SystemZISD::PACK, 4,
5698 { 4, 5, 6, 7, 12, 13, 14, 15, 20, 21, 22, 23, 28, 29, 30, 31 } },
5699 // VPKF
5700 { SystemZISD::PACK, 2,
5701 { 2, 3, 6, 7, 10, 11, 14, 15, 18, 19, 22, 23, 26, 27, 30, 31 } },
5702 // VPKH
5703 { SystemZISD::PACK, 1,
5704 { 1, 3, 5, 7, 9, 11, 13, 15, 17, 19, 21, 23, 25, 27, 29, 31 } },
5705 // VPDI V1, V2, 4 (low half of V1, high half of V2)
5706 { SystemZISD::PERMUTE_DWORDS, 4,
5707 { 8, 9, 10, 11, 12, 13, 14, 15, 16, 17, 18, 19, 20, 21, 22, 23 } },
5708 // VPDI V1, V2, 1 (high half of V1, low half of V2)
5709 { SystemZISD::PERMUTE_DWORDS, 1,
5710 { 0, 1, 2, 3, 4, 5, 6, 7, 24, 25, 26, 27, 28, 29, 30, 31 } }
5711};
5712
5713// Called after matching a vector shuffle against a particular pattern.
5714// Both the original shuffle and the pattern have two vector operands.
5715// OpNos[0] is the operand of the original shuffle that should be used for
5716// operand 0 of the pattern, or -1 if operand 0 of the pattern can be anything.
5717// OpNos[1] is the same for operand 1 of the pattern. Resolve these -1s and
5718// set OpNo0 and OpNo1 to the shuffle operands that should actually be used
5719// for operands 0 and 1 of the pattern.
5720static bool chooseShuffleOpNos(int *OpNos, unsigned &OpNo0, unsigned &OpNo1) {
5721 if (OpNos[0] < 0) {
5722 if (OpNos[1] < 0)
5723 return false;
5724 OpNo0 = OpNo1 = OpNos[1];
5725 } else if (OpNos[1] < 0) {
5726 OpNo0 = OpNo1 = OpNos[0];
5727 } else {
5728 OpNo0 = OpNos[0];
5729 OpNo1 = OpNos[1];
5730 }
5731 return true;
5732}
5733
5734// Bytes is a VPERM-like permute vector, except that -1 is used for
5735// undefined bytes. Return true if the VPERM can be implemented using P.
5736// When returning true set OpNo0 to the VPERM operand that should be
5737// used for operand 0 of P and likewise OpNo1 for operand 1 of P.
5738//
5739// For example, if swapping the VPERM operands allows P to match, OpNo0
5740// will be 1 and OpNo1 will be 0. If instead Bytes only refers to one
5741// operand, but rewriting it to use two duplicated operands allows it to
5742// match P, then OpNo0 and OpNo1 will be the same.
5743static bool matchPermute(const SmallVectorImpl<int> &Bytes, const Permute &P,
5744 unsigned &OpNo0, unsigned &OpNo1) {
5745 int OpNos[] = { -1, -1 };
5746 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5747 int Elt = Bytes[I];
5748 if (Elt >= 0) {
5749 // Make sure that the two permute vectors use the same suboperand
5750 // byte number. Only the operand numbers (the high bits) are
5751 // allowed to differ.
5752 if ((Elt ^ P.Bytes[I]) & (SystemZ::VectorBytes - 1))
5753 return false;
5754 int ModelOpNo = P.Bytes[I] / SystemZ::VectorBytes;
5755 int RealOpNo = unsigned(Elt) / SystemZ::VectorBytes;
5756 // Make sure that the operand mappings are consistent with previous
5757 // elements.
5758 if (OpNos[ModelOpNo] == 1 - RealOpNo)
5759 return false;
5760 OpNos[ModelOpNo] = RealOpNo;
5761 }
5762 }
5763 return chooseShuffleOpNos(OpNos, OpNo0, OpNo1);
5764}
5765
5766// As above, but search for a matching permute.
5767static const Permute *matchPermute(const SmallVectorImpl<int> &Bytes,
5768 unsigned &OpNo0, unsigned &OpNo1) {
5769 for (auto &P : PermuteForms)
5770 if (matchPermute(Bytes, P, OpNo0, OpNo1))
5771 return &P;
5772 return nullptr;
5773}
5774
5775// Bytes is a VPERM-like permute vector, except that -1 is used for
5776// undefined bytes. This permute is an operand of an outer permute.
5777// See whether redistributing the -1 bytes gives a shuffle that can be
5778// implemented using P. If so, set Transform to a VPERM-like permute vector
5779// that, when applied to the result of P, gives the original permute in Bytes.
5781 const Permute &P,
5782 SmallVectorImpl<int> &Transform) {
5783 unsigned To = 0;
5784 for (unsigned From = 0; From < SystemZ::VectorBytes; ++From) {
5785 int Elt = Bytes[From];
5786 if (Elt < 0)
5787 // Byte number From of the result is undefined.
5788 Transform[From] = -1;
5789 else {
5790 while (P.Bytes[To] != Elt) {
5791 To += 1;
5792 if (To == SystemZ::VectorBytes)
5793 return false;
5794 }
5795 Transform[From] = To;
5796 }
5797 }
5798 return true;
5799}
5800
5801// As above, but search for a matching permute.
5802static const Permute *matchDoublePermute(const SmallVectorImpl<int> &Bytes,
5803 SmallVectorImpl<int> &Transform) {
5804 for (auto &P : PermuteForms)
5805 if (matchDoublePermute(Bytes, P, Transform))
5806 return &P;
5807 return nullptr;
5808}
5809
5810// Convert the mask of the given shuffle op into a byte-level mask,
5811// as if it had type vNi8.
5812static bool getVPermMask(SDValue ShuffleOp,
5813 SmallVectorImpl<int> &Bytes) {
5814 EVT VT = ShuffleOp.getValueType();
5815 unsigned NumElements = VT.getVectorNumElements();
5816 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
5817
5818 if (auto *VSN = dyn_cast<ShuffleVectorSDNode>(ShuffleOp)) {
5819 Bytes.resize(NumElements * BytesPerElement, -1);
5820 for (unsigned I = 0; I < NumElements; ++I) {
5821 int Index = VSN->getMaskElt(I);
5822 if (Index >= 0)
5823 for (unsigned J = 0; J < BytesPerElement; ++J)
5824 Bytes[I * BytesPerElement + J] = Index * BytesPerElement + J;
5825 }
5826 return true;
5827 }
5828 if (SystemZISD::SPLAT == ShuffleOp.getOpcode() &&
5829 isa<ConstantSDNode>(ShuffleOp.getOperand(1))) {
5830 unsigned Index = ShuffleOp.getConstantOperandVal(1);
5831 Bytes.resize(NumElements * BytesPerElement, -1);
5832 for (unsigned I = 0; I < NumElements; ++I)
5833 for (unsigned J = 0; J < BytesPerElement; ++J)
5834 Bytes[I * BytesPerElement + J] = Index * BytesPerElement + J;
5835 return true;
5836 }
5837 return false;
5838}
5839
5840// Bytes is a VPERM-like permute vector, except that -1 is used for
5841// undefined bytes. See whether bytes [Start, Start + BytesPerElement) of
5842// the result come from a contiguous sequence of bytes from one input.
5843// Set Base to the selector for the first byte if so.
5844static bool getShuffleInput(const SmallVectorImpl<int> &Bytes, unsigned Start,
5845 unsigned BytesPerElement, int &Base) {
5846 Base = -1;
5847 for (unsigned I = 0; I < BytesPerElement; ++I) {
5848 if (Bytes[Start + I] >= 0) {
5849 unsigned Elem = Bytes[Start + I];
5850 if (Base < 0) {
5851 Base = Elem - I;
5852 // Make sure the bytes would come from one input operand.
5853 if (unsigned(Base) % Bytes.size() + BytesPerElement > Bytes.size())
5854 return false;
5855 } else if (unsigned(Base) != Elem - I)
5856 return false;
5857 }
5858 }
5859 return true;
5860}
5861
5862// Bytes is a VPERM-like permute vector, except that -1 is used for
5863// undefined bytes. Return true if it can be performed using VSLDB.
5864// When returning true, set StartIndex to the shift amount and OpNo0
5865// and OpNo1 to the VPERM operands that should be used as the first
5866// and second shift operand respectively.
5868 unsigned &StartIndex, unsigned &OpNo0,
5869 unsigned &OpNo1) {
5870 int OpNos[] = { -1, -1 };
5871 int Shift = -1;
5872 for (unsigned I = 0; I < 16; ++I) {
5873 int Index = Bytes[I];
5874 if (Index >= 0) {
5875 int ExpectedShift = (Index - I) % SystemZ::VectorBytes;
5876 int ModelOpNo = unsigned(ExpectedShift + I) / SystemZ::VectorBytes;
5877 int RealOpNo = unsigned(Index) / SystemZ::VectorBytes;
5878 if (Shift < 0)
5879 Shift = ExpectedShift;
5880 else if (Shift != ExpectedShift)
5881 return false;
5882 // Make sure that the operand mappings are consistent with previous
5883 // elements.
5884 if (OpNos[ModelOpNo] == 1 - RealOpNo)
5885 return false;
5886 OpNos[ModelOpNo] = RealOpNo;
5887 }
5888 }
5889 StartIndex = Shift;
5890 return chooseShuffleOpNos(OpNos, OpNo0, OpNo1);
5891}
5892
5893// Create a node that performs P on operands Op0 and Op1, casting the
5894// operands to the appropriate type. The type of the result is determined by P.
5896 const Permute &P, SDValue Op0, SDValue Op1) {
5897 // VPDI (PERMUTE_DWORDS) always operates on v2i64s. The input
5898 // elements of a PACK are twice as wide as the outputs.
5899 unsigned InBytes = (P.Opcode == SystemZISD::PERMUTE_DWORDS ? 8 :
5900 P.Opcode == SystemZISD::PACK ? P.Operand * 2 :
5901 P.Operand);
5902 // Cast both operands to the appropriate type.
5903 MVT InVT = MVT::getVectorVT(MVT::getIntegerVT(InBytes * 8),
5904 SystemZ::VectorBytes / InBytes);
5905 Op0 = DAG.getNode(ISD::BITCAST, DL, InVT, Op0);
5906 Op1 = DAG.getNode(ISD::BITCAST, DL, InVT, Op1);
5907 SDValue Op;
5908 if (P.Opcode == SystemZISD::PERMUTE_DWORDS) {
5909 SDValue Op2 = DAG.getTargetConstant(P.Operand, DL, MVT::i32);
5910 Op = DAG.getNode(SystemZISD::PERMUTE_DWORDS, DL, InVT, Op0, Op1, Op2);
5911 } else if (P.Opcode == SystemZISD::PACK) {
5912 MVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(P.Operand * 8),
5913 SystemZ::VectorBytes / P.Operand);
5914 Op = DAG.getNode(SystemZISD::PACK, DL, OutVT, Op0, Op1);
5915 } else {
5916 Op = DAG.getNode(P.Opcode, DL, InVT, Op0, Op1);
5917 }
5918 return Op;
5919}
5920
5921static bool isZeroVector(SDValue N) {
5922 if (N->getOpcode() == ISD::BITCAST)
5923 N = N->getOperand(0);
5924 if (N->getOpcode() == ISD::SPLAT_VECTOR)
5925 if (auto *Op = dyn_cast<ConstantSDNode>(N->getOperand(0)))
5926 return Op->getZExtValue() == 0;
5927 return ISD::isBuildVectorAllZeros(N.getNode());
5928}
5929
5930// Return the index of the zero/undef vector, or UINT32_MAX if not found.
5931static uint32_t findZeroVectorIdx(SDValue *Ops, unsigned Num) {
5932 for (unsigned I = 0; I < Num ; I++)
5933 if (isZeroVector(Ops[I]))
5934 return I;
5935 return UINT32_MAX;
5936}
5937
5938// Bytes is a VPERM-like permute vector, except that -1 is used for
5939// undefined bytes. Implement it on operands Ops[0] and Ops[1] using
5940// VSLDB or VPERM.
5942 SDValue *Ops,
5943 const SmallVectorImpl<int> &Bytes) {
5944 for (unsigned I = 0; I < 2; ++I)
5945 Ops[I] = DAG.getNode(ISD::BITCAST, DL, MVT::v16i8, Ops[I]);
5946
5947 // First see whether VSLDB can be used.
5948 unsigned StartIndex, OpNo0, OpNo1;
5949 if (isShlDoublePermute(Bytes, StartIndex, OpNo0, OpNo1))
5950 return DAG.getNode(SystemZISD::SHL_DOUBLE, DL, MVT::v16i8, Ops[OpNo0],
5951 Ops[OpNo1],
5952 DAG.getTargetConstant(StartIndex, DL, MVT::i32));
5953
5954 // Fall back on VPERM. Construct an SDNode for the permute vector. Try to
5955 // eliminate a zero vector by reusing any zero index in the permute vector.
5956 unsigned ZeroVecIdx = findZeroVectorIdx(&Ops[0], 2);
5957 if (ZeroVecIdx != UINT32_MAX) {
5958 bool MaskFirst = true;
5959 int ZeroIdx = -1;
5960 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5961 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
5962 unsigned Byte = unsigned(Bytes[I]) % SystemZ::VectorBytes;
5963 if (OpNo == ZeroVecIdx && I == 0) {
5964 // If the first byte is zero, use mask as first operand.
5965 ZeroIdx = 0;
5966 break;
5967 }
5968 if (OpNo != ZeroVecIdx && Byte == 0) {
5969 // If mask contains a zero, use it by placing that vector first.
5970 ZeroIdx = I + SystemZ::VectorBytes;
5971 MaskFirst = false;
5972 break;
5973 }
5974 }
5975 if (ZeroIdx != -1) {
5976 SDValue IndexNodes[SystemZ::VectorBytes];
5977 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I) {
5978 if (Bytes[I] >= 0) {
5979 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
5980 unsigned Byte = unsigned(Bytes[I]) % SystemZ::VectorBytes;
5981 if (OpNo == ZeroVecIdx)
5982 IndexNodes[I] = DAG.getConstant(ZeroIdx, DL, MVT::i32);
5983 else {
5984 unsigned BIdx = MaskFirst ? Byte + SystemZ::VectorBytes : Byte;
5985 IndexNodes[I] = DAG.getConstant(BIdx, DL, MVT::i32);
5986 }
5987 } else
5988 IndexNodes[I] = DAG.getUNDEF(MVT::i32);
5989 }
5990 SDValue Mask = DAG.getBuildVector(MVT::v16i8, DL, IndexNodes);
5991 SDValue Src = ZeroVecIdx == 0 ? Ops[1] : Ops[0];
5992 if (MaskFirst)
5993 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Mask, Src,
5994 Mask);
5995 else
5996 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Src, Mask,
5997 Mask);
5998 }
5999 }
6000
6001 SDValue IndexNodes[SystemZ::VectorBytes];
6002 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6003 if (Bytes[I] >= 0)
6004 IndexNodes[I] = DAG.getConstant(Bytes[I], DL, MVT::i32);
6005 else
6006 IndexNodes[I] = DAG.getUNDEF(MVT::i32);
6007 SDValue Op2 = DAG.getBuildVector(MVT::v16i8, DL, IndexNodes);
6008 return DAG.getNode(SystemZISD::PERMUTE, DL, MVT::v16i8, Ops[0],
6009 (!Ops[1].isUndef() ? Ops[1] : Ops[0]), Op2);
6010}
6011
6012namespace {
6013// Describes a general N-operand vector shuffle.
6014struct GeneralShuffle {
6015 GeneralShuffle(EVT vt)
6016 : VT(vt), UnpackFromEltSize(UINT_MAX), UnpackLow(false) {}
6017 void addUndef();
6018 bool add(SDValue, unsigned);
6019 SDValue getNode(SelectionDAG &, const SDLoc &);
6020 void tryPrepareForUnpack();
6021 bool unpackWasPrepared() { return UnpackFromEltSize <= 4; }
6022 SDValue insertUnpackIfPrepared(SelectionDAG &DAG, const SDLoc &DL, SDValue Op);
6023
6024 // The operands of the shuffle.
6026
6027 // Index I is -1 if byte I of the result is undefined. Otherwise the
6028 // result comes from byte Bytes[I] % SystemZ::VectorBytes of operand
6029 // Bytes[I] / SystemZ::VectorBytes.
6031
6032 // The type of the shuffle result.
6033 EVT VT;
6034
6035 // Holds a value of 1, 2 or 4 if a final unpack has been prepared for.
6036 unsigned UnpackFromEltSize;
6037 // True if the final unpack uses the low half.
6038 bool UnpackLow;
6039};
6040} // namespace
6041
6042// Add an extra undefined element to the shuffle.
6043void GeneralShuffle::addUndef() {
6044 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
6045 for (unsigned I = 0; I < BytesPerElement; ++I)
6046 Bytes.push_back(-1);
6047}
6048
6049// Add an extra element to the shuffle, taking it from element Elem of Op.
6050// A null Op indicates a vector input whose value will be calculated later;
6051// there is at most one such input per shuffle and it always has the same
6052// type as the result. Aborts and returns false if the source vector elements
6053// of an EXTRACT_VECTOR_ELT are smaller than the destination elements. Per
6054// LLVM they become implicitly extended, but this is rare and not optimized.
6055bool GeneralShuffle::add(SDValue Op, unsigned Elem) {
6056 unsigned BytesPerElement = VT.getVectorElementType().getStoreSize();
6057
6058 // The source vector can have wider elements than the result,
6059 // either through an explicit TRUNCATE or because of type legalization.
6060 // We want the least significant part.
6061 EVT FromVT = Op.getNode() ? Op.getValueType() : VT;
6062 unsigned FromBytesPerElement = FromVT.getVectorElementType().getStoreSize();
6063
6064 // Return false if the source elements are smaller than their destination
6065 // elements.
6066 if (FromBytesPerElement < BytesPerElement)
6067 return false;
6068
6069 unsigned Byte = ((Elem * FromBytesPerElement) % SystemZ::VectorBytes +
6070 (FromBytesPerElement - BytesPerElement));
6071
6072 // Look through things like shuffles and bitcasts.
6073 while (Op.getNode()) {
6074 if (Op.getOpcode() == ISD::BITCAST)
6075 Op = Op.getOperand(0);
6076 else if (Op.getOpcode() == ISD::VECTOR_SHUFFLE && Op.hasOneUse()) {
6077 // See whether the bytes we need come from a contiguous part of one
6078 // operand.
6080 if (!getVPermMask(Op, OpBytes))
6081 break;
6082 int NewByte;
6083 if (!getShuffleInput(OpBytes, Byte, BytesPerElement, NewByte))
6084 break;
6085 if (NewByte < 0) {
6086 addUndef();
6087 return true;
6088 }
6089 Op = Op.getOperand(unsigned(NewByte) / SystemZ::VectorBytes);
6090 Byte = unsigned(NewByte) % SystemZ::VectorBytes;
6091 } else if (Op.isUndef()) {
6092 addUndef();
6093 return true;
6094 } else
6095 break;
6096 }
6097
6098 // Make sure that the source of the extraction is in Ops.
6099 unsigned OpNo = 0;
6100 for (; OpNo < Ops.size(); ++OpNo)
6101 if (Ops[OpNo] == Op)
6102 break;
6103 if (OpNo == Ops.size())
6104 Ops.push_back(Op);
6105
6106 // Add the element to Bytes.
6107 unsigned Base = OpNo * SystemZ::VectorBytes + Byte;
6108 for (unsigned I = 0; I < BytesPerElement; ++I)
6109 Bytes.push_back(Base + I);
6110
6111 return true;
6112}
6113
6114// Return SDNodes for the completed shuffle.
6115SDValue GeneralShuffle::getNode(SelectionDAG &DAG, const SDLoc &DL) {
6116 assert(Bytes.size() == SystemZ::VectorBytes && "Incomplete vector");
6117
6118 if (Ops.size() == 0)
6119 return DAG.getUNDEF(VT);
6120
6121 // Use a single unpack if possible as the last operation.
6122 tryPrepareForUnpack();
6123
6124 // Make sure that there are at least two shuffle operands.
6125 if (Ops.size() == 1)
6126 Ops.push_back(DAG.getUNDEF(MVT::v16i8));
6127
6128 // Create a tree of shuffles, deferring root node until after the loop.
6129 // Try to redistribute the undefined elements of non-root nodes so that
6130 // the non-root shuffles match something like a pack or merge, then adjust
6131 // the parent node's permute vector to compensate for the new order.
6132 // Among other things, this copes with vectors like <2 x i16> that were
6133 // padded with undefined elements during type legalization.
6134 //
6135 // In the best case this redistribution will lead to the whole tree
6136 // using packs and merges. It should rarely be a loss in other cases.
6137 unsigned Stride = 1;
6138 for (; Stride * 2 < Ops.size(); Stride *= 2) {
6139 for (unsigned I = 0; I < Ops.size() - Stride; I += Stride * 2) {
6140 SDValue SubOps[] = { Ops[I], Ops[I + Stride] };
6141
6142 // Create a mask for just these two operands.
6144 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J) {
6145 unsigned OpNo = unsigned(Bytes[J]) / SystemZ::VectorBytes;
6146 unsigned Byte = unsigned(Bytes[J]) % SystemZ::VectorBytes;
6147 if (OpNo == I)
6148 NewBytes[J] = Byte;
6149 else if (OpNo == I + Stride)
6150 NewBytes[J] = SystemZ::VectorBytes + Byte;
6151 else
6152 NewBytes[J] = -1;
6153 }
6154 // See if it would be better to reorganize NewMask to avoid using VPERM.
6156 if (const Permute *P = matchDoublePermute(NewBytes, NewBytesMap)) {
6157 Ops[I] = getPermuteNode(DAG, DL, *P, SubOps[0], SubOps[1]);
6158 // Applying NewBytesMap to Ops[I] gets back to NewBytes.
6159 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J) {
6160 if (NewBytes[J] >= 0) {
6161 assert(unsigned(NewBytesMap[J]) < SystemZ::VectorBytes &&
6162 "Invalid double permute");
6163 Bytes[J] = I * SystemZ::VectorBytes + NewBytesMap[J];
6164 } else
6165 assert(NewBytesMap[J] < 0 && "Invalid double permute");
6166 }
6167 } else {
6168 // Just use NewBytes on the operands.
6169 Ops[I] = getGeneralPermuteNode(DAG, DL, SubOps, NewBytes);
6170 for (unsigned J = 0; J < SystemZ::VectorBytes; ++J)
6171 if (NewBytes[J] >= 0)
6172 Bytes[J] = I * SystemZ::VectorBytes + J;
6173 }
6174 }
6175 }
6176
6177 // Now we just have 2 inputs. Put the second operand in Ops[1].
6178 if (Stride > 1) {
6179 Ops[1] = Ops[Stride];
6180 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6181 if (Bytes[I] >= int(SystemZ::VectorBytes))
6182 Bytes[I] -= (Stride - 1) * SystemZ::VectorBytes;
6183 }
6184
6185 // Look for an instruction that can do the permute without resorting
6186 // to VPERM.
6187 unsigned OpNo0, OpNo1;
6188 SDValue Op;
6189 if (unpackWasPrepared() && Ops[1].isUndef())
6190 Op = Ops[0];
6191 else if (const Permute *P = matchPermute(Bytes, OpNo0, OpNo1))
6192 Op = getPermuteNode(DAG, DL, *P, Ops[OpNo0], Ops[OpNo1]);
6193 else
6194 Op = getGeneralPermuteNode(DAG, DL, &Ops[0], Bytes);
6195
6196 Op = insertUnpackIfPrepared(DAG, DL, Op);
6197
6198 return DAG.getNode(ISD::BITCAST, DL, VT, Op);
6199}
6200
6201#ifndef NDEBUG
6202static void dumpBytes(const SmallVectorImpl<int> &Bytes, std::string Msg) {
6203 dbgs() << Msg.c_str() << " { ";
6204 for (unsigned I = 0; I < Bytes.size(); I++)
6205 dbgs() << Bytes[I] << " ";
6206 dbgs() << "}\n";
6207}
6208#endif
6209
6210// If the Bytes vector matches an unpack operation, prepare to do the unpack
6211// after all else by removing the zero vector and the effect of the unpack on
6212// Bytes.
6213void GeneralShuffle::tryPrepareForUnpack() {
6214 uint32_t ZeroVecOpNo = findZeroVectorIdx(&Ops[0], Ops.size());
6215 if (ZeroVecOpNo == UINT32_MAX || Ops.size() == 1)
6216 return;
6217
6218 // Only do this if removing the zero vector reduces the depth, otherwise
6219 // the critical path will increase with the final unpack.
6220 if (Ops.size() > 2 &&
6221 Log2_32_Ceil(Ops.size()) == Log2_32_Ceil(Ops.size() - 1))
6222 return;
6223
6224 // Find an unpack that would allow removing the zero vector from Ops.
6225 UnpackFromEltSize = 1;
6226 for (; UnpackFromEltSize <= 4; UnpackFromEltSize *= 2) {
6227 bool MatchUnpack = true;
6229 for (unsigned Elt = 0; Elt < SystemZ::VectorBytes; Elt++) {
6230 unsigned ToEltSize = UnpackFromEltSize * 2;
6231 bool IsZextByte = (Elt % ToEltSize) < UnpackFromEltSize;
6232 if (!IsZextByte)
6233 SrcBytes.push_back(Bytes[Elt]);
6234 if (Bytes[Elt] != -1) {
6235 unsigned OpNo = unsigned(Bytes[Elt]) / SystemZ::VectorBytes;
6236 if (IsZextByte != (OpNo == ZeroVecOpNo)) {
6237 MatchUnpack = false;
6238 break;
6239 }
6240 }
6241 }
6242 if (MatchUnpack) {
6243 if (Ops.size() == 2) {
6244 // Don't use unpack if a single source operand needs rearrangement.
6245 bool CanUseUnpackLow = true, CanUseUnpackHigh = true;
6246 for (unsigned i = 0; i < SystemZ::VectorBytes / 2; i++) {
6247 if (SrcBytes[i] == -1)
6248 continue;
6249 if (SrcBytes[i] % 16 != int(i))
6250 CanUseUnpackHigh = false;
6251 if (SrcBytes[i] % 16 != int(i + SystemZ::VectorBytes / 2))
6252 CanUseUnpackLow = false;
6253 if (!CanUseUnpackLow && !CanUseUnpackHigh) {
6254 UnpackFromEltSize = UINT_MAX;
6255 return;
6256 }
6257 }
6258 if (!CanUseUnpackHigh)
6259 UnpackLow = true;
6260 }
6261 break;
6262 }
6263 }
6264 if (UnpackFromEltSize > 4)
6265 return;
6266
6267 LLVM_DEBUG(dbgs() << "Preparing for final unpack of element size "
6268 << UnpackFromEltSize << ". Zero vector is Op#" << ZeroVecOpNo
6269 << ".\n";
6270 dumpBytes(Bytes, "Original Bytes vector:"););
6271
6272 // Apply the unpack in reverse to the Bytes array.
6273 unsigned B = 0;
6274 if (UnpackLow) {
6275 while (B < SystemZ::VectorBytes / 2)
6276 Bytes[B++] = -1;
6277 }
6278 for (unsigned Elt = 0; Elt < SystemZ::VectorBytes;) {
6279 Elt += UnpackFromEltSize;
6280 for (unsigned i = 0; i < UnpackFromEltSize; i++, Elt++, B++)
6281 Bytes[B] = Bytes[Elt];
6282 }
6283 if (!UnpackLow) {
6284 while (B < SystemZ::VectorBytes)
6285 Bytes[B++] = -1;
6286 }
6287
6288 // Remove the zero vector from Ops
6289 Ops.erase(&Ops[ZeroVecOpNo]);
6290 for (unsigned I = 0; I < SystemZ::VectorBytes; ++I)
6291 if (Bytes[I] >= 0) {
6292 unsigned OpNo = unsigned(Bytes[I]) / SystemZ::VectorBytes;
6293 if (OpNo > ZeroVecOpNo)
6294 Bytes[I] -= SystemZ::VectorBytes;
6295 }
6296
6297 LLVM_DEBUG(dumpBytes(Bytes, "Resulting Bytes vector, zero vector removed:");
6298 dbgs() << "\n";);
6299}
6300
6301SDValue GeneralShuffle::insertUnpackIfPrepared(SelectionDAG &DAG,
6302 const SDLoc &DL,
6303 SDValue Op) {
6304 if (!unpackWasPrepared())
6305 return Op;
6306 unsigned InBits = UnpackFromEltSize * 8;
6307 EVT InVT = MVT::getVectorVT(MVT::getIntegerVT(InBits),
6308 SystemZ::VectorBits / InBits);
6309 SDValue PackedOp = DAG.getNode(ISD::BITCAST, DL, InVT, Op);
6310 unsigned OutBits = InBits * 2;
6311 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(OutBits),
6312 SystemZ::VectorBits / OutBits);
6313 return DAG.getNode(UnpackLow ? SystemZISD::UNPACKL_LOW
6314 : SystemZISD::UNPACKL_HIGH,
6315 DL, OutVT, PackedOp);
6316}
6317
6318// Return true if the given BUILD_VECTOR is a scalar-to-vector conversion.
6320 for (unsigned I = 1, E = Op.getNumOperands(); I != E; ++I)
6321 if (!Op.getOperand(I).isUndef())
6322 return false;
6323 return true;
6324}
6325
6326// Return a vector of type VT that contains Value in the first element.
6327// The other elements don't matter.
6329 SDValue Value) {
6330 // If we have a constant, replicate it to all elements and let the
6331 // BUILD_VECTOR lowering take care of it.
6332 if (Value.getOpcode() == ISD::Constant ||
6333 Value.getOpcode() == ISD::ConstantFP) {
6335 return DAG.getBuildVector(VT, DL, Ops);
6336 }
6337 if (Value.isUndef())
6338 return DAG.getUNDEF(VT);
6339 return DAG.getNode(ISD::SCALAR_TO_VECTOR, DL, VT, Value);
6340}
6341
6342// Return a vector of type VT in which Op0 is in element 0 and Op1 is in
6343// element 1. Used for cases in which replication is cheap.
6345 SDValue Op0, SDValue Op1) {
6346 if (Op0.isUndef()) {
6347 if (Op1.isUndef())
6348 return DAG.getUNDEF(VT);
6349 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op1);
6350 }
6351 if (Op1.isUndef())
6352 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op0);
6353 return DAG.getNode(SystemZISD::MERGE_HIGH, DL, VT,
6354 buildScalarToVector(DAG, DL, VT, Op0),
6355 buildScalarToVector(DAG, DL, VT, Op1));
6356}
6357
6358// Extend GPR scalars Op0 and Op1 to doublewords and return a v2i64
6359// vector for them.
6361 SDValue Op1) {
6362 if (Op0.isUndef() && Op1.isUndef())
6363 return DAG.getUNDEF(MVT::v2i64);
6364 // If one of the two inputs is undefined then replicate the other one,
6365 // in order to avoid using another register unnecessarily.
6366 if (Op0.isUndef())
6367 Op0 = Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op1);
6368 else if (Op1.isUndef())
6369 Op0 = Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
6370 else {
6371 Op0 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op0);
6372 Op1 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Op1);
6373 }
6374 return DAG.getNode(SystemZISD::JOIN_DWORDS, DL, MVT::v2i64, Op0, Op1);
6375}
6376
6377// If a BUILD_VECTOR contains some EXTRACT_VECTOR_ELTs, it's usually
6378// better to use VECTOR_SHUFFLEs on them, only using BUILD_VECTOR for
6379// the non-EXTRACT_VECTOR_ELT elements. See if the given BUILD_VECTOR
6380// would benefit from this representation and return it if so.
6382 BuildVectorSDNode *BVN) {
6383 EVT VT = BVN->getValueType(0);
6384 unsigned NumElements = VT.getVectorNumElements();
6385
6386 // Represent the BUILD_VECTOR as an N-operand VECTOR_SHUFFLE-like operation
6387 // on byte vectors. If there are non-EXTRACT_VECTOR_ELT elements that still
6388 // need a BUILD_VECTOR, add an additional placeholder operand for that
6389 // BUILD_VECTOR and store its operands in ResidueOps.
6390 GeneralShuffle GS(VT);
6392 bool FoundOne = false;
6393 for (unsigned I = 0; I < NumElements; ++I) {
6394 SDValue Op = BVN->getOperand(I);
6395 if (Op.getOpcode() == ISD::TRUNCATE)
6396 Op = Op.getOperand(0);
6397 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
6398 Op.getOperand(1).getOpcode() == ISD::Constant) {
6399 unsigned Elem = Op.getConstantOperandVal(1);
6400 if (!GS.add(Op.getOperand(0), Elem))
6401 return SDValue();
6402 FoundOne = true;
6403 } else if (Op.isUndef()) {
6404 GS.addUndef();
6405 } else {
6406 if (!GS.add(SDValue(), ResidueOps.size()))
6407 return SDValue();
6408 ResidueOps.push_back(BVN->getOperand(I));
6409 }
6410 }
6411
6412 // Nothing to do if there are no EXTRACT_VECTOR_ELTs.
6413 if (!FoundOne)
6414 return SDValue();
6415
6416 // Create the BUILD_VECTOR for the remaining elements, if any.
6417 if (!ResidueOps.empty()) {
6418 while (ResidueOps.size() < NumElements)
6419 ResidueOps.push_back(DAG.getUNDEF(ResidueOps[0].getValueType()));
6420 for (auto &Op : GS.Ops) {
6421 if (!Op.getNode()) {
6422 Op = DAG.getBuildVector(VT, SDLoc(BVN), ResidueOps);
6423 break;
6424 }
6425 }
6426 }
6427 return GS.getNode(DAG, SDLoc(BVN));
6428}
6429
6430bool SystemZTargetLowering::isVectorElementLoad(SDValue Op) const {
6431 if (Op.getOpcode() == ISD::LOAD && cast<LoadSDNode>(Op)->isUnindexed())
6432 return true;
6433 if (auto *AL = dyn_cast<AtomicSDNode>(Op))
6434 if (AL->getOpcode() == ISD::ATOMIC_LOAD)
6435 return true;
6436 if (Subtarget.hasVectorEnhancements2() && Op.getOpcode() == SystemZISD::LRV)
6437 return true;
6438 return false;
6439}
6440
6442 unsigned MergedBits, EVT VT, SDValue Op0,
6443 SDValue Op1) {
6444 MVT IntVecVT = MVT::getVectorVT(MVT::getIntegerVT(MergedBits),
6445 SystemZ::VectorBits / MergedBits);
6446 assert(VT.getSizeInBits() == 128 && IntVecVT.getSizeInBits() == 128 &&
6447 "Handling full vectors only.");
6448 Op0 = DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0);
6449 Op1 = DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op1);
6450 SDValue Op = DAG.getNode(SystemZISD::MERGE_HIGH, DL, IntVecVT, Op0, Op1);
6451 return DAG.getNode(ISD::BITCAST, DL, VT, Op);
6452}
6453
6455 EVT VT, SmallVectorImpl<SDValue> &Elems,
6456 unsigned Pos) {
6457 SDValue Op01 = buildMergeScalars(DAG, DL, VT, Elems[Pos + 0], Elems[Pos + 1]);
6458 SDValue Op23 = buildMergeScalars(DAG, DL, VT, Elems[Pos + 2], Elems[Pos + 3]);
6459 // Avoid unnecessary undefs by reusing the other operand.
6460 if (Op01.isUndef()) {
6461 if (Op23.isUndef())
6462 return Op01;
6463 Op01 = Op23;
6464 } else if (Op23.isUndef())
6465 Op23 = Op01;
6466 // Merging identical replications is a no-op.
6467 if (Op01.getOpcode() == SystemZISD::REPLICATE && Op01 == Op23)
6468 return Op01;
6469 unsigned MergedBits = VT.getSimpleVT().getScalarSizeInBits() * 2;
6470 return mergeHighParts(DAG, DL, MergedBits, VT, Op01, Op23);
6471}
6472
6473// Combine GPR scalar values Elems into a vector of type VT.
6474SDValue
6475SystemZTargetLowering::buildVector(SelectionDAG &DAG, const SDLoc &DL, EVT VT,
6476 SmallVectorImpl<SDValue> &Elems) const {
6477 // See whether there is a single replicated value.
6478 SDValue Single;
6479 unsigned int NumElements = Elems.size();
6480 unsigned int Count = 0;
6481 for (auto Elem : Elems) {
6482 if (!Elem.isUndef()) {
6483 if (!Single.getNode())
6484 Single = Elem;
6485 else if (Elem != Single) {
6486 Single = SDValue();
6487 break;
6488 }
6489 Count += 1;
6490 }
6491 }
6492 // There are three cases here:
6493 //
6494 // - if the only defined element is a loaded one, the best sequence
6495 // is a replicating load.
6496 //
6497 // - otherwise, if the only defined element is an i64 value, we will
6498 // end up with the same VLVGP sequence regardless of whether we short-cut
6499 // for replication or fall through to the later code.
6500 //
6501 // - otherwise, if the only defined element is an i32 or smaller value,
6502 // we would need 2 instructions to replicate it: VLVGP followed by VREPx.
6503 // This is only a win if the single defined element is used more than once.
6504 // In other cases we're better off using a single VLVGx.
6505 if (Single.getNode() && (Count > 1 || isVectorElementLoad(Single)))
6506 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Single);
6507
6508 // If all elements are loads, use VLREP/VLEs (below).
6509 bool AllLoads = true;
6510 for (auto Elem : Elems)
6511 if (!isVectorElementLoad(Elem)) {
6512 AllLoads = false;
6513 break;
6514 }
6515
6516 // The best way of building a v2i64 from two i64s is to use VLVGP.
6517 if (VT == MVT::v2i64 && !AllLoads)
6518 return joinDwords(DAG, DL, Elems[0], Elems[1]);
6519
6520 // Use a 64-bit merge high to combine two doubles.
6521 if (VT == MVT::v2f64 && !AllLoads)
6522 return buildMergeScalars(DAG, DL, VT, Elems[0], Elems[1]);
6523
6524 // Build v4f32 values directly from the FPRs:
6525 //
6526 // <Axxx> <Bxxx> <Cxxxx> <Dxxx>
6527 // V V VMRHF
6528 // <ABxx> <CDxx>
6529 // V VMRHG
6530 // <ABCD>
6531 if (VT == MVT::v4f32 && !AllLoads)
6532 return buildFPVecFromScalars4(DAG, DL, VT, Elems, 0);
6533
6534 // Same for v8f16.
6535 if (VT == MVT::v8f16 && !AllLoads) {
6536 SDValue Op0123 = buildFPVecFromScalars4(DAG, DL, VT, Elems, 0);
6537 SDValue Op4567 = buildFPVecFromScalars4(DAG, DL, VT, Elems, 4);
6538 // Avoid unnecessary undefs by reusing the other operand.
6539 if (Op0123.isUndef())
6540 Op0123 = Op4567;
6541 else if (Op4567.isUndef())
6542 Op4567 = Op0123;
6543 // Merging identical replications is a no-op.
6544 if (Op0123.getOpcode() == SystemZISD::REPLICATE && Op0123 == Op4567)
6545 return Op0123;
6546 return mergeHighParts(DAG, DL, 64, VT, Op0123, Op4567);
6547 }
6548
6549 // Collect the constant terms.
6552
6553 unsigned NumConstants = 0;
6554 for (unsigned I = 0; I < NumElements; ++I) {
6555 SDValue Elem = Elems[I];
6556 if (Elem.getOpcode() == ISD::Constant ||
6557 Elem.getOpcode() == ISD::ConstantFP) {
6558 NumConstants += 1;
6559 Constants[I] = Elem;
6560 Done[I] = true;
6561 }
6562 }
6563 // If there was at least one constant, fill in the other elements of
6564 // Constants with undefs to get a full vector constant and use that
6565 // as the starting point.
6566 SDValue Result;
6567 SDValue ReplicatedVal;
6568 if (NumConstants > 0) {
6569 for (unsigned I = 0; I < NumElements; ++I)
6570 if (!Constants[I].getNode())
6571 Constants[I] = DAG.getUNDEF(Elems[I].getValueType());
6572 Result = DAG.getBuildVector(VT, DL, Constants);
6573 } else {
6574 // Otherwise try to use VLREP or VLVGP to start the sequence in order to
6575 // avoid a false dependency on any previous contents of the vector
6576 // register.
6577
6578 // Use a VLREP if at least one element is a load. Make sure to replicate
6579 // the load with the most elements having its value.
6580 std::map<const SDNode*, unsigned> UseCounts;
6581 SDNode *LoadMaxUses = nullptr;
6582 for (unsigned I = 0; I < NumElements; ++I)
6583 if (isVectorElementLoad(Elems[I])) {
6584 SDNode *Ld = Elems[I].getNode();
6585 unsigned Count = ++UseCounts[Ld];
6586 if (LoadMaxUses == nullptr || UseCounts[LoadMaxUses] < Count)
6587 LoadMaxUses = Ld;
6588 }
6589 if (LoadMaxUses != nullptr) {
6590 ReplicatedVal = SDValue(LoadMaxUses, 0);
6591 Result = DAG.getNode(SystemZISD::REPLICATE, DL, VT, ReplicatedVal);
6592 } else {
6593 // Try to use VLVGP.
6594 unsigned I1 = NumElements / 2 - 1;
6595 unsigned I2 = NumElements - 1;
6596 bool Def1 = !Elems[I1].isUndef();
6597 bool Def2 = !Elems[I2].isUndef();
6598 if (Def1 || Def2) {
6599 SDValue Elem1 = Elems[Def1 ? I1 : I2];
6600 SDValue Elem2 = Elems[Def2 ? I2 : I1];
6601 Result = DAG.getNode(ISD::BITCAST, DL, VT,
6602 joinDwords(DAG, DL, Elem1, Elem2));
6603 Done[I1] = true;
6604 Done[I2] = true;
6605 } else
6606 Result = DAG.getUNDEF(VT);
6607 }
6608 }
6609
6610 // Use VLVGx to insert the other elements.
6611 for (unsigned I = 0; I < NumElements; ++I)
6612 if (!Done[I] && !Elems[I].isUndef() && Elems[I] != ReplicatedVal)
6613 Result = DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, VT, Result, Elems[I],
6614 DAG.getConstant(I, DL, MVT::i32));
6615 return Result;
6616}
6617
6618SDValue SystemZTargetLowering::lowerBUILD_VECTOR(SDValue Op,
6619 SelectionDAG &DAG) const {
6620 auto *BVN = cast<BuildVectorSDNode>(Op.getNode());
6621 SDLoc DL(Op);
6622 EVT VT = Op.getValueType();
6623
6624 if (BVN->isConstant()) {
6625 if (SystemZVectorConstantInfo(BVN).isVectorConstantLegal(Subtarget))
6626 return Op;
6627
6628 // Fall back to loading it from memory.
6629 return SDValue();
6630 }
6631
6632 // See if we should use shuffles to construct the vector from other vectors.
6633 if (SDValue Res = tryBuildVectorShuffle(DAG, BVN))
6634 return Res;
6635
6636 // Detect SCALAR_TO_VECTOR conversions.
6638 return buildScalarToVector(DAG, DL, VT, Op.getOperand(0));
6639
6640 // Otherwise use buildVector to build the vector up from GPRs.
6641 unsigned NumElements = Op.getNumOperands();
6643 for (unsigned I = 0; I < NumElements; ++I)
6644 Ops[I] = Op.getOperand(I);
6645 return buildVector(DAG, DL, VT, Ops);
6646}
6647
6648SDValue SystemZTargetLowering::lowerVECTOR_SHUFFLE(SDValue Op,
6649 SelectionDAG &DAG) const {
6650 auto *VSN = cast<ShuffleVectorSDNode>(Op.getNode());
6651 SDLoc DL(Op);
6652 EVT VT = Op.getValueType();
6653 unsigned NumElements = VT.getVectorNumElements();
6654
6655 if (VSN->isSplat()) {
6656 SDValue Op0 = Op.getOperand(0);
6657 unsigned Index = VSN->getSplatIndex();
6658 assert(Index < VT.getVectorNumElements() &&
6659 "Splat index should be defined and in first operand");
6660 // See whether the value we're splatting is directly available as a scalar.
6661 if ((Index == 0 && Op0.getOpcode() == ISD::SCALAR_TO_VECTOR) ||
6663 return DAG.getNode(SystemZISD::REPLICATE, DL, VT, Op0.getOperand(Index));
6664 // Otherwise keep it as a vector-to-vector operation.
6665 return DAG.getNode(SystemZISD::SPLAT, DL, VT, Op.getOperand(0),
6666 DAG.getTargetConstant(Index, DL, MVT::i32));
6667 }
6668
6669 GeneralShuffle GS(VT);
6670 for (unsigned I = 0; I < NumElements; ++I) {
6671 int Elt = VSN->getMaskElt(I);
6672 if (Elt < 0)
6673 GS.addUndef();
6674 else if (!GS.add(Op.getOperand(unsigned(Elt) / NumElements),
6675 unsigned(Elt) % NumElements))
6676 return SDValue();
6677 }
6678 return GS.getNode(DAG, SDLoc(VSN));
6679}
6680
6681SDValue SystemZTargetLowering::lowerSCALAR_TO_VECTOR(SDValue Op,
6682 SelectionDAG &DAG) const {
6683 SDLoc DL(Op);
6684 // Just insert the scalar into element 0 of an undefined vector.
6685 return DAG.getNode(ISD::INSERT_VECTOR_ELT, DL,
6686 Op.getValueType(), DAG.getUNDEF(Op.getValueType()),
6687 Op.getOperand(0), DAG.getConstant(0, DL, MVT::i32));
6688}
6689
6690// Shift the lower 2 bytes of Op to the left in order to insert into the
6691// upper 2 bytes of the FP register.
6693 assert(Op.getSimpleValueType() == MVT::i64 &&
6694 "Expexted to convert i64 to f16.");
6695 SDLoc DL(Op);
6696 SDValue Shft = DAG.getNode(ISD::SHL, DL, MVT::i64, Op,
6697 DAG.getConstant(48, DL, MVT::i64));
6698 SDValue BCast = DAG.getNode(ISD::BITCAST, DL, MVT::f64, Shft);
6699 SDValue F16Val =
6700 DAG.getTargetExtractSubreg(SystemZ::subreg_h16, DL, MVT::f16, BCast);
6701 return F16Val;
6702}
6703
6704// Extract Op into GPR and shift the 2 f16 bytes to the right.
6706 assert(Op.getSimpleValueType() == MVT::f16 &&
6707 "Expected to convert f16 to i64.");
6708 SDNode *U32 = DAG.getMachineNode(TargetOpcode::IMPLICIT_DEF, DL, MVT::f64);
6709 SDValue In64 = DAG.getTargetInsertSubreg(SystemZ::subreg_h16, DL, MVT::f64,
6710 SDValue(U32, 0), Op);
6711 SDValue BCast = DAG.getNode(ISD::BITCAST, DL, MVT::i64, In64);
6712 SDValue Shft = DAG.getNode(ISD::SRL, DL, MVT::i64, BCast,
6713 DAG.getConstant(48, DL, MVT::i32));
6714 return Shft;
6715}
6716
6717SDValue SystemZTargetLowering::lowerINSERT_VECTOR_ELT(SDValue Op,
6718 SelectionDAG &DAG) const {
6719 // Handle insertions of floating-point values.
6720 SDLoc DL(Op);
6721 SDValue Op0 = Op.getOperand(0);
6722 SDValue Op1 = Op.getOperand(1);
6723 SDValue Op2 = Op.getOperand(2);
6724 EVT VT = Op.getValueType();
6725
6726 // Insertions into constant indices of a v2f64 can be done using VPDI.
6727 // However, if the inserted value is a bitcast or a constant then it's
6728 // better to use GPRs, as below.
6729 if (VT == MVT::v2f64 &&
6730 Op1.getOpcode() != ISD::BITCAST &&
6731 Op1.getOpcode() != ISD::ConstantFP &&
6732 Op2.getOpcode() == ISD::Constant) {
6733 uint64_t Index = Op2->getAsZExtVal();
6734 unsigned Mask = VT.getVectorNumElements() - 1;
6735 if (Index <= Mask)
6736 return Op;
6737 }
6738
6739 // Otherwise bitcast to the equivalent integer form and insert via a GPR.
6740 MVT IntVT = MVT::getIntegerVT(VT.getScalarSizeInBits());
6741 MVT IntVecVT = MVT::getVectorVT(IntVT, VT.getVectorNumElements());
6742 SDValue IntOp1 =
6743 VT == MVT::v8f16
6744 ? DAG.getZExtOrTrunc(convertFromF16(Op1, DL, DAG), DL, MVT::i32)
6745 : DAG.getNode(ISD::BITCAST, DL, IntVT, Op1);
6746 SDValue Res =
6747 DAG.getNode(ISD::INSERT_VECTOR_ELT, DL, IntVecVT,
6748 DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0), IntOp1, Op2);
6749 return DAG.getNode(ISD::BITCAST, DL, VT, Res);
6750}
6751
6752SDValue
6753SystemZTargetLowering::lowerEXTRACT_VECTOR_ELT(SDValue Op,
6754 SelectionDAG &DAG) const {
6755 // Handle extractions of floating-point values.
6756 SDLoc DL(Op);
6757 SDValue Op0 = Op.getOperand(0);
6758 SDValue Op1 = Op.getOperand(1);
6759 EVT VT = Op.getValueType();
6760 EVT VecVT = Op0.getValueType();
6761
6762 // Extractions of constant indices can be done directly.
6763 if (auto *CIndexN = dyn_cast<ConstantSDNode>(Op1)) {
6764 uint64_t Index = CIndexN->getZExtValue();
6765 unsigned Mask = VecVT.getVectorNumElements() - 1;
6766 if (Index <= Mask)
6767 return Op;
6768 }
6769
6770 // Otherwise bitcast to the equivalent integer form and extract via a GPR.
6771 MVT IntVT = MVT::getIntegerVT(VT.getSizeInBits());
6772 MVT IntVecVT = MVT::getVectorVT(IntVT, VecVT.getVectorNumElements());
6773 MVT ExtrVT = IntVT == MVT::i16 ? MVT::i32 : IntVT;
6774 SDValue Extr = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ExtrVT,
6775 DAG.getNode(ISD::BITCAST, DL, IntVecVT, Op0), Op1);
6776 if (VT == MVT::f16)
6777 return convertToF16(DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Extr), DAG);
6778 return DAG.getNode(ISD::BITCAST, DL, VT, Extr);
6779}
6780
6781SDValue SystemZTargetLowering::
6782lowerSIGN_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const {
6783 SDValue PackedOp = Op.getOperand(0);
6784 EVT OutVT = Op.getValueType();
6785 EVT InVT = PackedOp.getValueType();
6786 unsigned ToBits = OutVT.getScalarSizeInBits();
6787 unsigned FromBits = InVT.getScalarSizeInBits();
6788 unsigned StartOffset = 0;
6789
6790 // If the input is a VECTOR_SHUFFLE, there are a number of important
6791 // cases where we can directly implement the sign-extension of the
6792 // original input lanes of the shuffle.
6793 if (PackedOp.getOpcode() == ISD::VECTOR_SHUFFLE) {
6794 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(PackedOp.getNode());
6795 ArrayRef<int> ShuffleMask = SVN->getMask();
6796 int OutNumElts = OutVT.getVectorNumElements();
6797
6798 // Recognize the special case where the sign-extension can be done
6799 // by the VSEG instruction. Handled via the default expander.
6800 if (ToBits == 64 && OutNumElts == 2) {
6801 int NumElem = ToBits / FromBits;
6802 if (ShuffleMask[0] == NumElem - 1 && ShuffleMask[1] == 2 * NumElem - 1)
6803 return SDValue();
6804 }
6805
6806 // Recognize the special case where we can fold the shuffle by
6807 // replacing some of the UNPACK_HIGH with UNPACK_LOW.
6808 int StartOffsetCandidate = -1;
6809 for (int Elt = 0; Elt < OutNumElts; Elt++) {
6810 if (ShuffleMask[Elt] == -1)
6811 continue;
6812 if (ShuffleMask[Elt] % OutNumElts == Elt) {
6813 if (StartOffsetCandidate == -1)
6814 StartOffsetCandidate = ShuffleMask[Elt] - Elt;
6815 if (StartOffsetCandidate == ShuffleMask[Elt] - Elt)
6816 continue;
6817 }
6818 StartOffsetCandidate = -1;
6819 break;
6820 }
6821 if (StartOffsetCandidate != -1) {
6822 StartOffset = StartOffsetCandidate;
6823 PackedOp = PackedOp.getOperand(0);
6824 }
6825 }
6826
6827 do {
6828 FromBits *= 2;
6829 unsigned OutNumElts = SystemZ::VectorBits / FromBits;
6830 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(FromBits), OutNumElts);
6831 unsigned Opcode = SystemZISD::UNPACK_HIGH;
6832 if (StartOffset >= OutNumElts) {
6833 Opcode = SystemZISD::UNPACK_LOW;
6834 StartOffset -= OutNumElts;
6835 }
6836 PackedOp = DAG.getNode(Opcode, SDLoc(PackedOp), OutVT, PackedOp);
6837 } while (FromBits != ToBits);
6838 return PackedOp;
6839}
6840
6841// Lower a ZERO_EXTEND_VECTOR_INREG to a vector shuffle with a zero vector.
6842SDValue SystemZTargetLowering::
6843lowerZERO_EXTEND_VECTOR_INREG(SDValue Op, SelectionDAG &DAG) const {
6844 SDValue PackedOp = Op.getOperand(0);
6845 SDLoc DL(Op);
6846 EVT OutVT = Op.getValueType();
6847 EVT InVT = PackedOp.getValueType();
6848 unsigned InNumElts = InVT.getVectorNumElements();
6849 unsigned OutNumElts = OutVT.getVectorNumElements();
6850 unsigned NumInPerOut = InNumElts / OutNumElts;
6851
6852 SDValue ZeroVec =
6853 DAG.getSplatVector(InVT, DL, DAG.getConstant(0, DL, InVT.getScalarType()));
6854
6855 SmallVector<int, 16> Mask(InNumElts);
6856 unsigned ZeroVecElt = InNumElts;
6857 for (unsigned PackedElt = 0; PackedElt < OutNumElts; PackedElt++) {
6858 unsigned MaskElt = PackedElt * NumInPerOut;
6859 unsigned End = MaskElt + NumInPerOut - 1;
6860 for (; MaskElt < End; MaskElt++)
6861 Mask[MaskElt] = ZeroVecElt++;
6862 Mask[MaskElt] = PackedElt;
6863 }
6864 SDValue Shuf = DAG.getVectorShuffle(InVT, DL, PackedOp, ZeroVec, Mask);
6865 return DAG.getNode(ISD::BITCAST, DL, OutVT, Shuf);
6866}
6867
6868SDValue SystemZTargetLowering::lowerShift(SDValue Op, SelectionDAG &DAG,
6869 unsigned ByScalar) const {
6870 // Look for cases where a vector shift can use the *_BY_SCALAR form.
6871 SDValue Op0 = Op.getOperand(0);
6872 SDValue Op1 = Op.getOperand(1);
6873 SDLoc DL(Op);
6874 EVT VT = Op.getValueType();
6875 unsigned ElemBitSize = VT.getScalarSizeInBits();
6876
6877 // See whether the shift vector is a splat represented as BUILD_VECTOR.
6878 if (auto *BVN = dyn_cast<BuildVectorSDNode>(Op1)) {
6879 APInt SplatBits, SplatUndef;
6880 unsigned SplatBitSize;
6881 bool HasAnyUndefs;
6882 // Check for constant splats. Use ElemBitSize as the minimum element
6883 // width and reject splats that need wider elements.
6884 if (BVN->isConstantSplat(SplatBits, SplatUndef, SplatBitSize, HasAnyUndefs,
6885 ElemBitSize, true) &&
6886 SplatBitSize == ElemBitSize) {
6887 SDValue Shift = DAG.getConstant(SplatBits.getZExtValue() & 0xfff,
6888 DL, MVT::i32);
6889 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6890 }
6891 // Check for variable splats.
6892 BitVector UndefElements;
6893 SDValue Splat = BVN->getSplatValue(&UndefElements);
6894 if (Splat) {
6895 // Since i32 is the smallest legal type, we either need a no-op
6896 // or a truncation.
6897 SDValue Shift = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32, Splat);
6898 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6899 }
6900 }
6901
6902 // See whether the shift vector is a splat represented as SHUFFLE_VECTOR,
6903 // and the shift amount is directly available in a GPR.
6904 if (auto *VSN = dyn_cast<ShuffleVectorSDNode>(Op1)) {
6905 if (VSN->isSplat()) {
6906 SDValue VSNOp0 = VSN->getOperand(0);
6907 unsigned Index = VSN->getSplatIndex();
6908 assert(Index < VT.getVectorNumElements() &&
6909 "Splat index should be defined and in first operand");
6910 if ((Index == 0 && VSNOp0.getOpcode() == ISD::SCALAR_TO_VECTOR) ||
6911 VSNOp0.getOpcode() == ISD::BUILD_VECTOR) {
6912 // Since i32 is the smallest legal type, we either need a no-op
6913 // or a truncation.
6914 SDValue Shift = DAG.getNode(ISD::TRUNCATE, DL, MVT::i32,
6915 VSNOp0.getOperand(Index));
6916 return DAG.getNode(ByScalar, DL, VT, Op0, Shift);
6917 }
6918 }
6919 }
6920
6921 // Otherwise just treat the current form as legal.
6922 return Op;
6923}
6924
6925SDValue SystemZTargetLowering::lowerFSHL(SDValue Op, SelectionDAG &DAG) const {
6926 SDLoc DL(Op);
6927
6928 // i128 FSHL with a constant amount that is a multiple of 8 can be
6929 // implemented via VECTOR_SHUFFLE. If we have the vector-enhancements-2
6930 // facility, FSHL with a constant amount less than 8 can be implemented
6931 // via SHL_DOUBLE_BIT, and FSHL with other constant amounts by a
6932 // combination of the two.
6933 if (auto *ShiftAmtNode = dyn_cast<ConstantSDNode>(Op.getOperand(2))) {
6934 uint64_t ShiftAmt = ShiftAmtNode->getZExtValue() & 127;
6935 if ((ShiftAmt & 7) == 0 || Subtarget.hasVectorEnhancements2()) {
6936 SDValue Op0 = DAG.getBitcast(MVT::v16i8, Op.getOperand(0));
6937 SDValue Op1 = DAG.getBitcast(MVT::v16i8, Op.getOperand(1));
6938 if (ShiftAmt > 120) {
6939 // For N in 121..128, fshl N == fshr (128 - N), and for 1 <= N < 8
6940 // SHR_DOUBLE_BIT emits fewer instructions.
6941 SDValue Val =
6942 DAG.getNode(SystemZISD::SHR_DOUBLE_BIT, DL, MVT::v16i8, Op0, Op1,
6943 DAG.getTargetConstant(128 - ShiftAmt, DL, MVT::i32));
6944 return DAG.getBitcast(MVT::i128, Val);
6945 }
6946 SmallVector<int, 16> Mask(16);
6947 for (unsigned Elt = 0; Elt < 16; Elt++)
6948 Mask[Elt] = (ShiftAmt >> 3) + Elt;
6949 SDValue Shuf1 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op1, Mask);
6950 if ((ShiftAmt & 7) == 0)
6951 return DAG.getBitcast(MVT::i128, Shuf1);
6952 SDValue Shuf2 = DAG.getVectorShuffle(MVT::v16i8, DL, Op1, Op1, Mask);
6953 SDValue Val =
6954 DAG.getNode(SystemZISD::SHL_DOUBLE_BIT, DL, MVT::v16i8, Shuf1, Shuf2,
6955 DAG.getTargetConstant(ShiftAmt & 7, DL, MVT::i32));
6956 return DAG.getBitcast(MVT::i128, Val);
6957 }
6958 }
6959
6960 return SDValue();
6961}
6962
6963SDValue SystemZTargetLowering::lowerFSHR(SDValue Op, SelectionDAG &DAG) const {
6964 SDLoc DL(Op);
6965
6966 // i128 FSHR with a constant amount that is a multiple of 8 can be
6967 // implemented via VECTOR_SHUFFLE. If we have the vector-enhancements-2
6968 // facility, FSHR with a constant amount less than 8 can be implemented
6969 // via SHR_DOUBLE_BIT, and FSHR with other constant amounts by a
6970 // combination of the two.
6971 if (auto *ShiftAmtNode = dyn_cast<ConstantSDNode>(Op.getOperand(2))) {
6972 uint64_t ShiftAmt = ShiftAmtNode->getZExtValue() & 127;
6973 if ((ShiftAmt & 7) == 0 || Subtarget.hasVectorEnhancements2()) {
6974 SDValue Op0 = DAG.getBitcast(MVT::v16i8, Op.getOperand(0));
6975 SDValue Op1 = DAG.getBitcast(MVT::v16i8, Op.getOperand(1));
6976 if (ShiftAmt > 120) {
6977 // For N in 121..128, fshr N == fshl (128 - N), and for 1 <= N < 8
6978 // SHL_DOUBLE_BIT emits fewer instructions.
6979 SDValue Val =
6980 DAG.getNode(SystemZISD::SHL_DOUBLE_BIT, DL, MVT::v16i8, Op0, Op1,
6981 DAG.getTargetConstant(128 - ShiftAmt, DL, MVT::i32));
6982 return DAG.getBitcast(MVT::i128, Val);
6983 }
6984 SmallVector<int, 16> Mask(16);
6985 for (unsigned Elt = 0; Elt < 16; Elt++)
6986 Mask[Elt] = 16 - (ShiftAmt >> 3) + Elt;
6987 SDValue Shuf1 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op1, Mask);
6988 if ((ShiftAmt & 7) == 0)
6989 return DAG.getBitcast(MVT::i128, Shuf1);
6990 SDValue Shuf2 = DAG.getVectorShuffle(MVT::v16i8, DL, Op0, Op0, Mask);
6991 SDValue Val =
6992 DAG.getNode(SystemZISD::SHR_DOUBLE_BIT, DL, MVT::v16i8, Shuf2, Shuf1,
6993 DAG.getTargetConstant(ShiftAmt & 7, DL, MVT::i32));
6994 return DAG.getBitcast(MVT::i128, Val);
6995 }
6996 }
6997
6998 return SDValue();
6999}
7000
7002 SDLoc DL(Op);
7003 SDValue Src = Op.getOperand(0);
7004 MVT DstVT = Op.getSimpleValueType();
7005
7007 unsigned SrcAS = N->getSrcAddressSpace();
7008
7009 assert(SrcAS != N->getDestAddressSpace() &&
7010 "addrspacecast must be between different address spaces");
7011
7012 // addrspacecast [0 <- 1] : Assinging a ptr32 value to a 64-bit pointer.
7013 // addrspacecast [1 <- 0] : Assigining a 64-bit pointer to a ptr32 value.
7014 if (SrcAS == SYSTEMZAS::PTR32 && DstVT == MVT::i64) {
7015 Op = DAG.getNode(ISD::AND, DL, MVT::i32, Src,
7016 DAG.getConstant(0x7fffffff, DL, MVT::i32));
7017 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, DstVT, Op);
7018 } else if (DstVT == MVT::i32) {
7019 Op = DAG.getNode(ISD::TRUNCATE, DL, DstVT, Src);
7020 Op = DAG.getNode(ISD::AND, DL, MVT::i32, Op,
7021 DAG.getConstant(0x7fffffff, DL, MVT::i32));
7022 Op = DAG.getNode(ISD::ZERO_EXTEND, DL, DstVT, Op);
7023 } else {
7024 report_fatal_error("Bad address space in addrspacecast");
7025 }
7026 return Op;
7027}
7028
7029SDValue SystemZTargetLowering::lowerFP_EXTEND(SDValue Op,
7030 SelectionDAG &DAG) const {
7031 SDValue In = Op.getOperand(Op->isStrictFPOpcode() ? 1 : 0);
7032 if (In.getSimpleValueType() != MVT::f16)
7033 return Op; // Legal
7034 return SDValue(); // Let legalizer emit the libcall.
7035}
7036
7038 MVT VT, SDValue Arg, SDLoc DL,
7039 SDValue Chain, bool IsStrict) const {
7040 assert(LC != RTLIB::UNKNOWN_LIBCALL && "Unexpected request for libcall!");
7041 MakeLibCallOptions CallOptions;
7042 SDValue Result;
7043 std::tie(Result, Chain) =
7044 makeLibCall(DAG, LC, VT, Arg, CallOptions, DL, Chain);
7045 return IsStrict ? DAG.getMergeValues({Result, Chain}, DL) : Result;
7046}
7047
7048SDValue SystemZTargetLowering::lower_FP_TO_INT(SDValue Op,
7049 SelectionDAG &DAG) const {
7050 bool IsSigned = (Op->getOpcode() == ISD::FP_TO_SINT ||
7051 Op->getOpcode() == ISD::STRICT_FP_TO_SINT);
7052 bool IsStrict = Op->isStrictFPOpcode();
7053 SDLoc DL(Op);
7054 MVT VT = Op.getSimpleValueType();
7055 SDValue InOp = Op.getOperand(IsStrict ? 1 : 0);
7056 SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
7057 EVT InVT = InOp.getValueType();
7058
7059 // FP to unsigned is not directly supported on z10. Promoting an i32
7060 // result to (signed) i64 doesn't generate an inexact condition (fp
7061 // exception) for values that are outside the i32 range but in the i64
7062 // range, so use the default expansion.
7063 if (!Subtarget.hasFPExtension() && !IsSigned)
7064 // Expand i32/i64. F16 values will be recognized to fit and extended.
7065 return SDValue();
7066
7067 // Conversion from f16 is done via f32.
7068 if (InOp.getSimpleValueType() == MVT::f16) {
7070 LowerOperationWrapper(Op.getNode(), Results, DAG);
7071 return DAG.getMergeValues(Results, DL);
7072 }
7073
7074 if (VT == MVT::i128) {
7075 RTLIB::Libcall LC =
7076 IsSigned ? RTLIB::getFPTOSINT(InVT, VT) : RTLIB::getFPTOUINT(InVT, VT);
7077 return useLibCall(DAG, LC, VT, InOp, DL, Chain, IsStrict);
7078 }
7079
7080 return Op; // Legal
7081}
7082
7083SDValue SystemZTargetLowering::lower_INT_TO_FP(SDValue Op,
7084 SelectionDAG &DAG) const {
7085 bool IsSigned = (Op->getOpcode() == ISD::SINT_TO_FP ||
7086 Op->getOpcode() == ISD::STRICT_SINT_TO_FP);
7087 bool IsStrict = Op->isStrictFPOpcode();
7088 SDLoc DL(Op);
7089 MVT VT = Op.getSimpleValueType();
7090 SDValue InOp = Op.getOperand(IsStrict ? 1 : 0);
7091 SDValue Chain = IsStrict ? Op.getOperand(0) : DAG.getEntryNode();
7092 EVT InVT = InOp.getValueType();
7093
7094 // Conversion to f16 is done via f32.
7095 if (VT == MVT::f16) {
7097 LowerOperationWrapper(Op.getNode(), Results, DAG);
7098 return DAG.getMergeValues(Results, DL);
7099 }
7100
7101 // Unsigned to fp is not directly supported on z10.
7102 if (!Subtarget.hasFPExtension() && !IsSigned)
7103 return SDValue(); // Expand i64.
7104
7105 if (InVT == MVT::i128) {
7106 RTLIB::Libcall LC =
7107 IsSigned ? RTLIB::getSINTTOFP(InVT, VT) : RTLIB::getUINTTOFP(InVT, VT);
7108 return useLibCall(DAG, LC, VT, InOp, DL, Chain, IsStrict);
7109 }
7110
7111 return Op; // Legal
7112}
7113
7114// Lower an f16 LOAD in case of no vector support.
7115SDValue SystemZTargetLowering::lowerLoadF16(SDValue Op,
7116 SelectionDAG &DAG) const {
7117 EVT RegVT = Op.getValueType();
7118 assert(RegVT == MVT::f16 && "Expected to lower an f16 load.");
7119 (void)RegVT;
7120
7121 // Load as integer.
7122 SDLoc DL(Op);
7123 SDValue NewLd;
7124 if (auto *AtomicLd = dyn_cast<AtomicSDNode>(Op.getNode())) {
7125 assert(EVT(RegVT) == AtomicLd->getMemoryVT() && "Unhandled f16 load");
7126 NewLd = DAG.getAtomicLoad(ISD::EXTLOAD, DL, MVT::i16, MVT::i64,
7127 AtomicLd->getChain(), AtomicLd->getBasePtr(),
7128 AtomicLd->getMemOperand());
7129 } else {
7130 LoadSDNode *Ld = cast<LoadSDNode>(Op.getNode());
7131 assert(EVT(RegVT) == Ld->getMemoryVT() && "Unhandled f16 load");
7132 NewLd = DAG.getExtLoad(ISD::EXTLOAD, DL, MVT::i64, Ld->getChain(),
7133 Ld->getBasePtr(), Ld->getPointerInfo(), MVT::i16,
7134 Ld->getBaseAlign(), Ld->getMemOperand()->getFlags());
7135 }
7136 SDValue F16Val = convertToF16(NewLd, DAG);
7137 return DAG.getMergeValues({F16Val, NewLd.getValue(1)}, DL);
7138}
7139
7140// Lower an f16 STORE in case of no vector support.
7141SDValue SystemZTargetLowering::lowerStoreF16(SDValue Op,
7142 SelectionDAG &DAG) const {
7143 SDLoc DL(Op);
7144 SDValue Shft = convertFromF16(Op->getOperand(1), DL, DAG);
7145
7146 if (auto *AtomicSt = dyn_cast<AtomicSDNode>(Op.getNode()))
7147 return DAG.getAtomic(ISD::ATOMIC_STORE, DL, MVT::i16, AtomicSt->getChain(),
7148 Shft, AtomicSt->getBasePtr(),
7149 AtomicSt->getMemOperand());
7150
7151 StoreSDNode *St = cast<StoreSDNode>(Op.getNode());
7152 return DAG.getTruncStore(St->getChain(), DL, Shft, St->getBasePtr(), MVT::i16,
7153 St->getMemOperand());
7154}
7155
7156SDValue SystemZTargetLowering::lowerIS_FPCLASS(SDValue Op,
7157 SelectionDAG &DAG) const {
7158 SDLoc DL(Op);
7159 MVT ResultVT = Op.getSimpleValueType();
7160 SDValue Arg = Op.getOperand(0);
7161 unsigned Check = Op.getConstantOperandVal(1);
7162
7163 unsigned TDCMask = 0;
7164 if (Check & fcSNan)
7166 if (Check & fcQNan)
7168 if (Check & fcPosInf)
7170 if (Check & fcNegInf)
7172 if (Check & fcPosNormal)
7174 if (Check & fcNegNormal)
7176 if (Check & fcPosSubnormal)
7178 if (Check & fcNegSubnormal)
7180 if (Check & fcPosZero)
7181 TDCMask |= SystemZ::TDCMASK_ZERO_PLUS;
7182 if (Check & fcNegZero)
7183 TDCMask |= SystemZ::TDCMASK_ZERO_MINUS;
7184 SDValue TDCMaskV = DAG.getConstant(TDCMask, DL, MVT::i64);
7185
7186 SDValue Intr = DAG.getNode(SystemZISD::TDC, DL, ResultVT, Arg, TDCMaskV);
7187 return getCCResult(DAG, Intr);
7188}
7189
7190SDValue SystemZTargetLowering::lowerREADCYCLECOUNTER(SDValue Op,
7191 SelectionDAG &DAG) const {
7192 SDLoc DL(Op);
7193 SDValue Chain = Op.getOperand(0);
7194
7195 // STCKF only supports a memory operand, so we have to use a temporary.
7196 SDValue StackPtr = DAG.CreateStackTemporary(MVT::i64);
7197 int SPFI = cast<FrameIndexSDNode>(StackPtr.getNode())->getIndex();
7198 MachinePointerInfo MPI =
7200
7201 // Use STCFK to store the TOD clock into the temporary.
7202 SDValue StoreOps[] = {Chain, StackPtr};
7203 Chain = DAG.getMemIntrinsicNode(
7204 SystemZISD::STCKF, DL, DAG.getVTList(MVT::Other), StoreOps, MVT::i64,
7205 MPI, MaybeAlign(), MachineMemOperand::MOStore);
7206
7207 // And read it back from there.
7208 return DAG.getLoad(MVT::i64, DL, Chain, StackPtr, MPI);
7209}
7210
7212 SelectionDAG &DAG) const {
7213 switch (Op.getOpcode()) {
7214 case ISD::FRAMEADDR:
7215 return lowerFRAMEADDR(Op, DAG);
7216 case ISD::RETURNADDR:
7217 return lowerRETURNADDR(Op, DAG);
7218 case ISD::BR_CC:
7219 return lowerBR_CC(Op, DAG);
7220 case ISD::SELECT_CC:
7221 return lowerSELECT_CC(Op, DAG);
7222 case ISD::SETCC:
7223 return lowerSETCC(Op, DAG);
7224 case ISD::STRICT_FSETCC:
7225 return lowerSTRICT_FSETCC(Op, DAG, false);
7227 return lowerSTRICT_FSETCC(Op, DAG, true);
7228 case ISD::GlobalAddress:
7229 return lowerGlobalAddress(cast<GlobalAddressSDNode>(Op), DAG);
7231 return lowerGlobalTLSAddress(cast<GlobalAddressSDNode>(Op), DAG);
7232 case ISD::BlockAddress:
7233 return lowerBlockAddress(cast<BlockAddressSDNode>(Op), DAG);
7234 case ISD::JumpTable:
7235 return lowerJumpTable(cast<JumpTableSDNode>(Op), DAG);
7236 case ISD::ConstantPool:
7237 return lowerConstantPool(cast<ConstantPoolSDNode>(Op), DAG);
7238 case ISD::BITCAST:
7239 return lowerBITCAST(Op, DAG);
7240 case ISD::VASTART:
7241 return lowerVASTART(Op, DAG);
7242 case ISD::VACOPY:
7243 return lowerVACOPY(Op, DAG);
7245 return lowerDYNAMIC_STACKALLOC(Op, DAG);
7247 return lowerGET_DYNAMIC_AREA_OFFSET(Op, DAG);
7248 case ISD::MULHS:
7249 return lowerMULH(Op, DAG, SystemZISD::SMUL_LOHI);
7250 case ISD::MULHU:
7251 return lowerMULH(Op, DAG, SystemZISD::UMUL_LOHI);
7252 case ISD::SMUL_LOHI:
7253 return lowerSMUL_LOHI(Op, DAG);
7254 case ISD::UMUL_LOHI:
7255 return lowerUMUL_LOHI(Op, DAG);
7256 case ISD::SDIVREM:
7257 return lowerSDIVREM(Op, DAG);
7258 case ISD::UDIVREM:
7259 return lowerUDIVREM(Op, DAG);
7260 case ISD::SADDO:
7261 case ISD::SSUBO:
7262 case ISD::UADDO:
7263 case ISD::USUBO:
7264 return lowerXALUO(Op, DAG);
7265 case ISD::UADDO_CARRY:
7266 case ISD::USUBO_CARRY:
7267 return lowerUADDSUBO_CARRY(Op, DAG);
7268 case ISD::OR:
7269 return lowerOR(Op, DAG);
7270 case ISD::CTPOP:
7271 return lowerCTPOP(Op, DAG);
7272 case ISD::VECREDUCE_ADD:
7273 return lowerVECREDUCE_ADD(Op, DAG);
7274 case ISD::ATOMIC_FENCE:
7275 return lowerATOMIC_FENCE(Op, DAG);
7276 case ISD::ATOMIC_SWAP:
7277 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_SWAPW);
7278 case ISD::ATOMIC_STORE:
7279 return lowerATOMIC_STORE(Op, DAG);
7280 case ISD::ATOMIC_LOAD:
7281 return lowerATOMIC_LOAD(Op, DAG);
7283 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_ADD);
7285 return lowerATOMIC_LOAD_SUB(Op, DAG);
7287 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_AND);
7289 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_OR);
7291 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_XOR);
7293 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_NAND);
7295 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_MIN);
7297 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_MAX);
7299 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_UMIN);
7301 return lowerATOMIC_LOAD_OP(Op, DAG, SystemZISD::ATOMIC_LOADW_UMAX);
7303 return lowerATOMIC_CMP_SWAP(Op, DAG);
7304 case ISD::STACKSAVE:
7305 return lowerSTACKSAVE(Op, DAG);
7306 case ISD::STACKRESTORE:
7307 return lowerSTACKRESTORE(Op, DAG);
7308 case ISD::PREFETCH:
7309 return lowerPREFETCH(Op, DAG);
7311 return lowerINTRINSIC_W_CHAIN(Op, DAG);
7313 return lowerINTRINSIC_WO_CHAIN(Op, DAG);
7314 case ISD::BUILD_VECTOR:
7315 return lowerBUILD_VECTOR(Op, DAG);
7317 return lowerVECTOR_SHUFFLE(Op, DAG);
7319 return lowerSCALAR_TO_VECTOR(Op, DAG);
7321 return lowerINSERT_VECTOR_ELT(Op, DAG);
7323 return lowerEXTRACT_VECTOR_ELT(Op, DAG);
7325 return lowerSIGN_EXTEND_VECTOR_INREG(Op, DAG);
7327 return lowerZERO_EXTEND_VECTOR_INREG(Op, DAG);
7328 case ISD::SHL:
7329 return lowerShift(Op, DAG, SystemZISD::VSHL_BY_SCALAR);
7330 case ISD::SRL:
7331 return lowerShift(Op, DAG, SystemZISD::VSRL_BY_SCALAR);
7332 case ISD::SRA:
7333 return lowerShift(Op, DAG, SystemZISD::VSRA_BY_SCALAR);
7334 case ISD::ADDRSPACECAST:
7335 return lowerAddrSpaceCast(Op, DAG);
7336 case ISD::ROTL:
7337 return lowerShift(Op, DAG, SystemZISD::VROTL_BY_SCALAR);
7338 case ISD::FSHL:
7339 return lowerFSHL(Op, DAG);
7340 case ISD::FSHR:
7341 return lowerFSHR(Op, DAG);
7342 case ISD::FP_EXTEND:
7344 return lowerFP_EXTEND(Op, DAG);
7345 case ISD::FP_TO_UINT:
7346 case ISD::FP_TO_SINT:
7349 return lower_FP_TO_INT(Op, DAG);
7350 case ISD::UINT_TO_FP:
7351 case ISD::SINT_TO_FP:
7354 return lower_INT_TO_FP(Op, DAG);
7355 case ISD::LOAD:
7356 return lowerLoadF16(Op, DAG);
7357 case ISD::STORE:
7358 return lowerStoreF16(Op, DAG);
7359 case ISD::IS_FPCLASS:
7360 return lowerIS_FPCLASS(Op, DAG);
7361 case ISD::GET_ROUNDING:
7362 return lowerGET_ROUNDING(Op, DAG);
7364 return lowerREADCYCLECOUNTER(Op, DAG);
7367 // These operations are legal on our platform, but we cannot actually
7368 // set the operation action to Legal as common code would treat this
7369 // as equivalent to Expand. Instead, we keep the operation action to
7370 // Custom and just leave them unchanged here.
7371 return Op;
7372
7373 default:
7374 llvm_unreachable("Unexpected node to lower");
7375 }
7376}
7377
7379 const SDLoc &SL) {
7380 // If i128 is legal, just use a normal bitcast.
7381 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128))
7382 return DAG.getBitcast(MVT::f128, Src);
7383
7384 // Otherwise, f128 must live in FP128, so do a partwise move.
7386 &SystemZ::FP128BitRegClass);
7387
7388 SDValue Hi, Lo;
7389 std::tie(Lo, Hi) = DAG.SplitScalar(Src, SL, MVT::i64, MVT::i64);
7390
7391 Hi = DAG.getBitcast(MVT::f64, Hi);
7392 Lo = DAG.getBitcast(MVT::f64, Lo);
7393
7394 SDNode *Pair = DAG.getMachineNode(
7395 SystemZ::REG_SEQUENCE, SL, MVT::f128,
7396 {DAG.getTargetConstant(SystemZ::FP128BitRegClassID, SL, MVT::i32), Lo,
7397 DAG.getTargetConstant(SystemZ::subreg_l64, SL, MVT::i32), Hi,
7398 DAG.getTargetConstant(SystemZ::subreg_h64, SL, MVT::i32)});
7399 return SDValue(Pair, 0);
7400}
7401
7403 const SDLoc &SL) {
7404 // If i128 is legal, just use a normal bitcast.
7405 if (DAG.getTargetLoweringInfo().isTypeLegal(MVT::i128))
7406 return DAG.getBitcast(MVT::i128, Src);
7407
7408 // Otherwise, f128 must live in FP128, so do a partwise move.
7410 &SystemZ::FP128BitRegClass);
7411
7412 SDValue LoFP =
7413 DAG.getTargetExtractSubreg(SystemZ::subreg_l64, SL, MVT::f64, Src);
7414 SDValue HiFP =
7415 DAG.getTargetExtractSubreg(SystemZ::subreg_h64, SL, MVT::f64, Src);
7416 SDValue Lo = DAG.getNode(ISD::BITCAST, SL, MVT::i64, LoFP);
7417 SDValue Hi = DAG.getNode(ISD::BITCAST, SL, MVT::i64, HiFP);
7418
7419 return DAG.getNode(ISD::BUILD_PAIR, SL, MVT::i128, Lo, Hi);
7420}
7421
7422// Lower operations with invalid operand or result types.
7423void
7426 SelectionDAG &DAG) const {
7427 switch (N->getOpcode()) {
7428 case ISD::ATOMIC_LOAD: {
7429 SDLoc DL(N);
7430 SDVTList Tys = DAG.getVTList(MVT::Untyped, MVT::Other);
7431 SDValue Ops[] = { N->getOperand(0), N->getOperand(1) };
7432 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7433 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_LOAD_128,
7434 DL, Tys, Ops, MVT::i128, MMO);
7435
7436 SDValue Lowered = lowerGR128ToI128(DAG, Res);
7437 if (N->getValueType(0) == MVT::f128)
7438 Lowered = expandBitCastI128ToF128(DAG, Lowered, DL);
7439 Results.push_back(Lowered);
7440 Results.push_back(Res.getValue(1));
7441 break;
7442 }
7443 case ISD::ATOMIC_STORE: {
7444 SDLoc DL(N);
7445 SDVTList Tys = DAG.getVTList(MVT::Other);
7446 SDValue Val = N->getOperand(1);
7447 if (Val.getValueType() == MVT::f128)
7448 Val = expandBitCastF128ToI128(DAG, Val, DL);
7449 Val = lowerI128ToGR128(DAG, Val);
7450
7451 SDValue Ops[] = {N->getOperand(0), Val, N->getOperand(2)};
7452 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7453 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_STORE_128,
7454 DL, Tys, Ops, MVT::i128, MMO);
7455 // We have to enforce sequential consistency by performing a
7456 // serialization operation after the store.
7457 if (cast<AtomicSDNode>(N)->getSuccessOrdering() ==
7459 Res = SDValue(DAG.getMachineNode(SystemZ::Serialize, DL,
7460 MVT::Other, Res), 0);
7461 Results.push_back(Res);
7462 break;
7463 }
7465 SDLoc DL(N);
7466 SDVTList Tys = DAG.getVTList(MVT::Untyped, MVT::i32, MVT::Other);
7467 SDValue Ops[] = { N->getOperand(0), N->getOperand(1),
7468 lowerI128ToGR128(DAG, N->getOperand(2)),
7469 lowerI128ToGR128(DAG, N->getOperand(3)) };
7470 MachineMemOperand *MMO = cast<AtomicSDNode>(N)->getMemOperand();
7471 SDValue Res = DAG.getMemIntrinsicNode(SystemZISD::ATOMIC_CMP_SWAP_128,
7472 DL, Tys, Ops, MVT::i128, MMO);
7473 SDValue Success = emitSETCC(DAG, DL, Res.getValue(1),
7475 Success = DAG.getZExtOrTrunc(Success, DL, N->getValueType(1));
7476 Results.push_back(lowerGR128ToI128(DAG, Res));
7477 Results.push_back(Success);
7478 Results.push_back(Res.getValue(2));
7479 break;
7480 }
7481 case ISD::BITCAST: {
7482 if (useSoftFloat())
7483 return;
7484 SDLoc DL(N);
7485 SDValue Src = N->getOperand(0);
7486 EVT SrcVT = Src.getValueType();
7487 EVT ResVT = N->getValueType(0);
7488 if (ResVT == MVT::i128 && SrcVT == MVT::f128)
7489 Results.push_back(expandBitCastF128ToI128(DAG, Src, DL));
7490 else if (SrcVT == MVT::i16 && ResVT == MVT::f16) {
7491 if (Subtarget.hasVector()) {
7492 SDValue In32 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i32, Src);
7493 Results.push_back(SDValue(
7494 DAG.getMachineNode(SystemZ::LEFR_16, DL, MVT::f16, In32), 0));
7495 } else {
7496 SDValue In64 = DAG.getNode(ISD::ANY_EXTEND, DL, MVT::i64, Src);
7497 Results.push_back(convertToF16(In64, DAG));
7498 }
7499 } else if (SrcVT == MVT::f16 && ResVT == MVT::i16) {
7500 SDValue ExtractedVal =
7501 Subtarget.hasVector()
7502 ? SDValue(DAG.getMachineNode(SystemZ::LFER_16, DL, MVT::i32, Src),
7503 0)
7504 : convertFromF16(Src, DL, DAG);
7505 Results.push_back(DAG.getZExtOrTrunc(ExtractedVal, DL, ResVT));
7506 }
7507 break;
7508 }
7509 case ISD::UINT_TO_FP:
7510 case ISD::SINT_TO_FP:
7513 if (useSoftFloat())
7514 return;
7515 bool IsStrict = N->isStrictFPOpcode();
7516 SDLoc DL(N);
7517 SDValue InOp = N->getOperand(IsStrict ? 1 : 0);
7518 EVT ResVT = N->getValueType(0);
7519 SDValue Chain = IsStrict ? N->getOperand(0) : DAG.getEntryNode();
7520 if (ResVT == MVT::f16) {
7521 if (!IsStrict) {
7522 SDValue OpF32 = DAG.getNode(N->getOpcode(), DL, MVT::f32, InOp);
7523 Results.push_back(DAG.getFPExtendOrRound(OpF32, DL, MVT::f16));
7524 } else {
7525 SDValue OpF32 =
7526 DAG.getNode(N->getOpcode(), DL, DAG.getVTList(MVT::f32, MVT::Other),
7527 {Chain, InOp});
7528 SDValue F16Res;
7529 std::tie(F16Res, Chain) = DAG.getStrictFPExtendOrRound(
7530 OpF32, OpF32.getValue(1), DL, MVT::f16);
7531 Results.push_back(F16Res);
7532 Results.push_back(Chain);
7533 }
7534 }
7535 break;
7536 }
7537 case ISD::FP_TO_UINT:
7538 case ISD::FP_TO_SINT:
7541 if (useSoftFloat())
7542 return;
7543 bool IsStrict = N->isStrictFPOpcode();
7544 SDLoc DL(N);
7545 EVT ResVT = N->getValueType(0);
7546 SDValue InOp = N->getOperand(IsStrict ? 1 : 0);
7547 EVT InVT = InOp->getValueType(0);
7548 SDValue Chain = IsStrict ? N->getOperand(0) : DAG.getEntryNode();
7549 if (InVT == MVT::f16) {
7550 if (!IsStrict) {
7551 SDValue InF32 = DAG.getFPExtendOrRound(InOp, DL, MVT::f32);
7552 Results.push_back(DAG.getNode(N->getOpcode(), DL, ResVT, InF32));
7553 } else {
7554 SDValue InF32;
7555 std::tie(InF32, Chain) =
7556 DAG.getStrictFPExtendOrRound(InOp, Chain, DL, MVT::f32);
7557 SDValue OpF32 =
7558 DAG.getNode(N->getOpcode(), DL, DAG.getVTList(ResVT, MVT::Other),
7559 {Chain, InF32});
7560 Results.push_back(OpF32);
7561 Results.push_back(OpF32.getValue(1));
7562 }
7563 }
7564 break;
7565 }
7566 default:
7567 llvm_unreachable("Unexpected node to lower");
7568 }
7569}
7570
7571void
7577
7578// Return true if VT is a vector whose elements are a whole number of bytes
7579// in width. Also check for presence of vector support.
7580bool SystemZTargetLowering::canTreatAsByteVector(EVT VT) const {
7581 if (!Subtarget.hasVector())
7582 return false;
7583
7584 return VT.isVector() && VT.getScalarSizeInBits() % 8 == 0 && VT.isSimple();
7585}
7586
7587// Try to simplify an EXTRACT_VECTOR_ELT from a vector of type VecVT
7588// producing a result of type ResVT. Op is a possibly bitcast version
7589// of the input vector and Index is the index (based on type VecVT) that
7590// should be extracted. Return the new extraction if a simplification
7591// was possible or if Force is true.
7592SDValue SystemZTargetLowering::combineExtract(const SDLoc &DL, EVT ResVT,
7593 EVT VecVT, SDValue Op,
7594 unsigned Index,
7595 DAGCombinerInfo &DCI,
7596 bool Force) const {
7597 SelectionDAG &DAG = DCI.DAG;
7598
7599 // The number of bytes being extracted.
7600 unsigned BytesPerElement = VecVT.getVectorElementType().getStoreSize();
7601
7602 for (;;) {
7603 unsigned Opcode = Op.getOpcode();
7604 if (Opcode == ISD::BITCAST)
7605 // Look through bitcasts.
7606 Op = Op.getOperand(0);
7607 else if ((Opcode == ISD::VECTOR_SHUFFLE || Opcode == SystemZISD::SPLAT) &&
7608 canTreatAsByteVector(Op.getValueType())) {
7609 // Get a VPERM-like permute mask and see whether the bytes covered
7610 // by the extracted element are a contiguous sequence from one
7611 // source operand.
7613 if (!getVPermMask(Op, Bytes))
7614 break;
7615 int First;
7616 if (!getShuffleInput(Bytes, Index * BytesPerElement,
7617 BytesPerElement, First))
7618 break;
7619 if (First < 0)
7620 return DAG.getUNDEF(ResVT);
7621 // Make sure the contiguous sequence starts at a multiple of the
7622 // original element size.
7623 unsigned Byte = unsigned(First) % Bytes.size();
7624 if (Byte % BytesPerElement != 0)
7625 break;
7626 // We can get the extracted value directly from an input.
7627 Index = Byte / BytesPerElement;
7628 Op = Op.getOperand(unsigned(First) / Bytes.size());
7629 Force = true;
7630 } else if (Opcode == ISD::BUILD_VECTOR &&
7631 canTreatAsByteVector(Op.getValueType())) {
7632 // We can only optimize this case if the BUILD_VECTOR elements are
7633 // at least as wide as the extracted value.
7634 EVT OpVT = Op.getValueType();
7635 unsigned OpBytesPerElement = OpVT.getVectorElementType().getStoreSize();
7636 if (OpBytesPerElement < BytesPerElement)
7637 break;
7638 // Make sure that the least-significant bit of the extracted value
7639 // is the least significant bit of an input.
7640 unsigned End = (Index + 1) * BytesPerElement;
7641 if (End % OpBytesPerElement != 0)
7642 break;
7643 // We're extracting the low part of one operand of the BUILD_VECTOR.
7644 Op = Op.getOperand(End / OpBytesPerElement - 1);
7645 EVT ResIntVT = MVT::getIntegerVT(ResVT.getSizeInBits());
7646 if (!isTypeLegal(ResIntVT))
7647 break;
7648 if (!Op.getValueType().isInteger()) {
7649 EVT OpIntVT = MVT::getIntegerVT(Op.getValueSizeInBits());
7650 if (!isTypeLegal(OpIntVT))
7651 break;
7652 Op = DAG.getNode(ISD::BITCAST, DL, OpIntVT, Op);
7653 DCI.AddToWorklist(Op.getNode());
7654 }
7655 Op = DAG.getNode(ISD::TRUNCATE, DL, ResIntVT, Op);
7656 if (ResIntVT != ResVT) {
7657 DCI.AddToWorklist(Op.getNode());
7658 Op = DAG.getNode(ISD::BITCAST, DL, ResVT, Op);
7659 }
7660 return Op;
7661 } else if ((Opcode == ISD::SIGN_EXTEND_VECTOR_INREG ||
7663 Opcode == ISD::ANY_EXTEND_VECTOR_INREG) &&
7664 canTreatAsByteVector(Op.getValueType()) &&
7665 canTreatAsByteVector(Op.getOperand(0).getValueType())) {
7666 // Make sure that only the unextended bits are significant.
7667 EVT ExtVT = Op.getValueType();
7668 EVT OpVT = Op.getOperand(0).getValueType();
7669 unsigned ExtBytesPerElement = ExtVT.getVectorElementType().getStoreSize();
7670 unsigned OpBytesPerElement = OpVT.getVectorElementType().getStoreSize();
7671 unsigned Byte = Index * BytesPerElement;
7672 unsigned SubByte = Byte % ExtBytesPerElement;
7673 unsigned MinSubByte = ExtBytesPerElement - OpBytesPerElement;
7674 if (SubByte < MinSubByte ||
7675 SubByte + BytesPerElement > ExtBytesPerElement)
7676 break;
7677 // Get the byte offset of the unextended element
7678 Byte = Byte / ExtBytesPerElement * OpBytesPerElement;
7679 // ...then add the byte offset relative to that element.
7680 Byte += SubByte - MinSubByte;
7681 if (Byte % BytesPerElement != 0)
7682 break;
7683 Op = Op.getOperand(0);
7684 Index = Byte / BytesPerElement;
7685 Force = true;
7686 } else
7687 break;
7688 }
7689 if (Force) {
7690 if (Op.getValueType() != VecVT) {
7691 Op = DAG.getNode(ISD::BITCAST, DL, VecVT, Op);
7692 DCI.AddToWorklist(Op.getNode());
7693 }
7694 return DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, ResVT, Op,
7695 DAG.getConstant(Index, DL, MVT::i32));
7696 }
7697 return SDValue();
7698}
7699
7700// Optimize vector operations in scalar value Op on the basis that Op
7701// is truncated to TruncVT.
7702SDValue SystemZTargetLowering::combineTruncateExtract(
7703 const SDLoc &DL, EVT TruncVT, SDValue Op, DAGCombinerInfo &DCI) const {
7704 // If we have (trunc (extract_vector_elt X, Y)), try to turn it into
7705 // (extract_vector_elt (bitcast X), Y'), where (bitcast X) has elements
7706 // of type TruncVT.
7707 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
7708 TruncVT.getSizeInBits() % 8 == 0) {
7709 SDValue Vec = Op.getOperand(0);
7710 EVT VecVT = Vec.getValueType();
7711 if (canTreatAsByteVector(VecVT)) {
7712 if (auto *IndexN = dyn_cast<ConstantSDNode>(Op.getOperand(1))) {
7713 unsigned BytesPerElement = VecVT.getVectorElementType().getStoreSize();
7714 unsigned TruncBytes = TruncVT.getStoreSize();
7715 if (BytesPerElement % TruncBytes == 0) {
7716 // Calculate the value of Y' in the above description. We are
7717 // splitting the original elements into Scale equal-sized pieces
7718 // and for truncation purposes want the last (least-significant)
7719 // of these pieces for IndexN. This is easiest to do by calculating
7720 // the start index of the following element and then subtracting 1.
7721 unsigned Scale = BytesPerElement / TruncBytes;
7722 unsigned NewIndex = (IndexN->getZExtValue() + 1) * Scale - 1;
7723
7724 // Defer the creation of the bitcast from X to combineExtract,
7725 // which might be able to optimize the extraction.
7726 VecVT = EVT::getVectorVT(*DCI.DAG.getContext(),
7727 MVT::getIntegerVT(TruncBytes * 8),
7728 VecVT.getStoreSize() / TruncBytes);
7729 EVT ResVT = (TruncBytes < 4 ? MVT::i32 : TruncVT);
7730 return combineExtract(DL, ResVT, VecVT, Vec, NewIndex, DCI, true);
7731 }
7732 }
7733 }
7734 }
7735 return SDValue();
7736}
7737
7738SDValue SystemZTargetLowering::combineZERO_EXTEND(
7739 SDNode *N, DAGCombinerInfo &DCI) const {
7740 // Convert (zext (select_ccmask C1, C2)) into (select_ccmask C1', C2')
7741 SelectionDAG &DAG = DCI.DAG;
7742 SDValue N0 = N->getOperand(0);
7743 EVT VT = N->getValueType(0);
7744 if (N0.getOpcode() == SystemZISD::SELECT_CCMASK) {
7745 auto *TrueOp = dyn_cast<ConstantSDNode>(N0.getOperand(0));
7746 auto *FalseOp = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7747 if (TrueOp && FalseOp) {
7748 SDLoc DL(N0);
7749 SDValue Ops[] = { DAG.getConstant(TrueOp->getZExtValue(), DL, VT),
7750 DAG.getConstant(FalseOp->getZExtValue(), DL, VT),
7751 N0.getOperand(2), N0.getOperand(3), N0.getOperand(4) };
7752 SDValue NewSelect = DAG.getNode(SystemZISD::SELECT_CCMASK, DL, VT, Ops);
7753 // If N0 has multiple uses, change other uses as well.
7754 if (!N0.hasOneUse()) {
7755 SDValue TruncSelect =
7756 DAG.getNode(ISD::TRUNCATE, DL, N0.getValueType(), NewSelect);
7757 DCI.CombineTo(N0.getNode(), TruncSelect);
7758 }
7759 return NewSelect;
7760 }
7761 }
7762 // Convert (zext (xor (trunc X), C)) into (xor (trunc X), C') if the size
7763 // of the result is smaller than the size of X and all the truncated bits
7764 // of X are already zero.
7765 if (N0.getOpcode() == ISD::XOR &&
7766 N0.hasOneUse() && N0.getOperand(0).hasOneUse() &&
7767 N0.getOperand(0).getOpcode() == ISD::TRUNCATE &&
7768 N0.getOperand(1).getOpcode() == ISD::Constant) {
7769 SDValue X = N0.getOperand(0).getOperand(0);
7770 if (VT.isScalarInteger() && VT.getSizeInBits() < X.getValueSizeInBits()) {
7771 KnownBits Known = DAG.computeKnownBits(X);
7772 APInt TruncatedBits = APInt::getBitsSet(X.getValueSizeInBits(),
7773 N0.getValueSizeInBits(),
7774 VT.getSizeInBits());
7775 if (TruncatedBits.isSubsetOf(Known.Zero)) {
7776 X = DAG.getNode(ISD::TRUNCATE, SDLoc(X), VT, X);
7777 APInt Mask = N0.getConstantOperandAPInt(1).zext(VT.getSizeInBits());
7778 return DAG.getNode(ISD::XOR, SDLoc(N0), VT,
7779 X, DAG.getConstant(Mask, SDLoc(N0), VT));
7780 }
7781 }
7782 }
7783 // Recognize patterns for VECTOR SUBTRACT COMPUTE BORROW INDICATION
7784 // and VECTOR ADD COMPUTE CARRY for i128:
7785 // (zext (setcc_uge X Y)) --> (VSCBI X Y)
7786 // (zext (setcc_ule Y X)) --> (VSCBI X Y)
7787 // (zext (setcc_ult (add X Y) X/Y) -> (VACC X Y)
7788 // (zext (setcc_ugt X/Y (add X Y)) -> (VACC X Y)
7789 // For vector types, these patterns are recognized in the .td file.
7790 if (N0.getOpcode() == ISD::SETCC && isTypeLegal(VT) && VT == MVT::i128 &&
7791 N0.getOperand(0).getValueType() == VT) {
7792 SDValue Op0 = N0.getOperand(0);
7793 SDValue Op1 = N0.getOperand(1);
7794 const ISD::CondCode CC = cast<CondCodeSDNode>(N0.getOperand(2))->get();
7795 switch (CC) {
7796 case ISD::SETULE:
7797 std::swap(Op0, Op1);
7798 [[fallthrough]];
7799 case ISD::SETUGE:
7800 return DAG.getNode(SystemZISD::VSCBI, SDLoc(N0), VT, Op0, Op1);
7801 case ISD::SETUGT:
7802 std::swap(Op0, Op1);
7803 [[fallthrough]];
7804 case ISD::SETULT:
7805 if (Op0->hasOneUse() && Op0->getOpcode() == ISD::ADD &&
7806 (Op0->getOperand(0) == Op1 || Op0->getOperand(1) == Op1))
7807 return DAG.getNode(SystemZISD::VACC, SDLoc(N0), VT, Op0->getOperand(0),
7808 Op0->getOperand(1));
7809 break;
7810 default:
7811 break;
7812 }
7813 }
7814
7815 return SDValue();
7816}
7817
7818SDValue SystemZTargetLowering::combineSIGN_EXTEND_INREG(
7819 SDNode *N, DAGCombinerInfo &DCI) const {
7820 // Convert (sext_in_reg (setcc LHS, RHS, COND), i1)
7821 // and (sext_in_reg (any_extend (setcc LHS, RHS, COND)), i1)
7822 // into (select_cc LHS, RHS, -1, 0, COND)
7823 SelectionDAG &DAG = DCI.DAG;
7824 SDValue N0 = N->getOperand(0);
7825 EVT VT = N->getValueType(0);
7826 EVT EVT = cast<VTSDNode>(N->getOperand(1))->getVT();
7827 if (N0.hasOneUse() && N0.getOpcode() == ISD::ANY_EXTEND)
7828 N0 = N0.getOperand(0);
7829 if (EVT == MVT::i1 && N0.hasOneUse() && N0.getOpcode() == ISD::SETCC) {
7830 SDLoc DL(N0);
7831 SDValue Ops[] = { N0.getOperand(0), N0.getOperand(1),
7832 DAG.getAllOnesConstant(DL, VT),
7833 DAG.getConstant(0, DL, VT), N0.getOperand(2) };
7834 return DAG.getNode(ISD::SELECT_CC, DL, VT, Ops);
7835 }
7836 return SDValue();
7837}
7838
7839SDValue SystemZTargetLowering::combineSIGN_EXTEND(
7840 SDNode *N, DAGCombinerInfo &DCI) const {
7841 // Convert (sext (ashr (shl X, C1), C2)) to
7842 // (ashr (shl (anyext X), C1'), C2')), since wider shifts are as
7843 // cheap as narrower ones.
7844 SelectionDAG &DAG = DCI.DAG;
7845 SDValue N0 = N->getOperand(0);
7846 EVT VT = N->getValueType(0);
7847 if (N0.hasOneUse() && N0.getOpcode() == ISD::SRA) {
7848 auto *SraAmt = dyn_cast<ConstantSDNode>(N0.getOperand(1));
7849 SDValue Inner = N0.getOperand(0);
7850 if (SraAmt && Inner.hasOneUse() && Inner.getOpcode() == ISD::SHL) {
7851 if (auto *ShlAmt = dyn_cast<ConstantSDNode>(Inner.getOperand(1))) {
7852 unsigned Extra = (VT.getSizeInBits() - N0.getValueSizeInBits());
7853 unsigned NewShlAmt = ShlAmt->getZExtValue() + Extra;
7854 unsigned NewSraAmt = SraAmt->getZExtValue() + Extra;
7855 EVT ShiftVT = N0.getOperand(1).getValueType();
7856 SDValue Ext = DAG.getNode(ISD::ANY_EXTEND, SDLoc(Inner), VT,
7857 Inner.getOperand(0));
7858 SDValue Shl = DAG.getNode(ISD::SHL, SDLoc(Inner), VT, Ext,
7859 DAG.getConstant(NewShlAmt, SDLoc(Inner),
7860 ShiftVT));
7861 return DAG.getNode(ISD::SRA, SDLoc(N0), VT, Shl,
7862 DAG.getConstant(NewSraAmt, SDLoc(N0), ShiftVT));
7863 }
7864 }
7865 }
7866
7867 return SDValue();
7868}
7869
7870SDValue SystemZTargetLowering::combineMERGE(
7871 SDNode *N, DAGCombinerInfo &DCI) const {
7872 SelectionDAG &DAG = DCI.DAG;
7873 unsigned Opcode = N->getOpcode();
7874 SDValue Op0 = N->getOperand(0);
7875 SDValue Op1 = N->getOperand(1);
7876 if (Op0.getOpcode() == ISD::BITCAST)
7877 Op0 = Op0.getOperand(0);
7879 // (z_merge_* 0, 0) -> 0. This is mostly useful for using VLLEZF
7880 // for v4f32.
7881 if (Op1 == N->getOperand(0))
7882 return Op1;
7883 // (z_merge_? 0, X) -> (z_unpackl_? 0, X).
7884 EVT VT = Op1.getValueType();
7885 unsigned ElemBytes = VT.getVectorElementType().getStoreSize();
7886 if (ElemBytes <= 4) {
7887 Opcode = (Opcode == SystemZISD::MERGE_HIGH ?
7888 SystemZISD::UNPACKL_HIGH : SystemZISD::UNPACKL_LOW);
7889 EVT InVT = VT.changeVectorElementTypeToInteger();
7890 EVT OutVT = MVT::getVectorVT(MVT::getIntegerVT(ElemBytes * 16),
7891 SystemZ::VectorBytes / ElemBytes / 2);
7892 if (VT != InVT) {
7893 Op1 = DAG.getNode(ISD::BITCAST, SDLoc(N), InVT, Op1);
7894 DCI.AddToWorklist(Op1.getNode());
7895 }
7896 SDValue Op = DAG.getNode(Opcode, SDLoc(N), OutVT, Op1);
7897 DCI.AddToWorklist(Op.getNode());
7898 return DAG.getNode(ISD::BITCAST, SDLoc(N), VT, Op);
7899 }
7900 }
7901 return SDValue();
7902}
7903
7904static bool isI128MovedToParts(LoadSDNode *LD, SDNode *&LoPart,
7905 SDNode *&HiPart) {
7906 LoPart = HiPart = nullptr;
7907
7908 // Scan through all users.
7909 for (SDUse &Use : LD->uses()) {
7910 // Skip the uses of the chain.
7911 if (Use.getResNo() != 0)
7912 continue;
7913
7914 // Verify every user is a TRUNCATE to i64 of the low or high half.
7915 SDNode *User = Use.getUser();
7916 bool IsLoPart = true;
7917 if (User->getOpcode() == ISD::SRL &&
7918 User->getOperand(1).getOpcode() == ISD::Constant &&
7919 User->getConstantOperandVal(1) == 64 && User->hasOneUse()) {
7920 User = *User->user_begin();
7921 IsLoPart = false;
7922 }
7923 if (User->getOpcode() != ISD::TRUNCATE || User->getValueType(0) != MVT::i64)
7924 return false;
7925
7926 if (IsLoPart) {
7927 if (LoPart)
7928 return false;
7929 LoPart = User;
7930 } else {
7931 if (HiPart)
7932 return false;
7933 HiPart = User;
7934 }
7935 }
7936 return true;
7937}
7938
7939static bool isF128MovedToParts(LoadSDNode *LD, SDNode *&LoPart,
7940 SDNode *&HiPart) {
7941 LoPart = HiPart = nullptr;
7942
7943 // Scan through all users.
7944 for (SDUse &Use : LD->uses()) {
7945 // Skip the uses of the chain.
7946 if (Use.getResNo() != 0)
7947 continue;
7948
7949 // Verify every user is an EXTRACT_SUBREG of the low or high half.
7950 SDNode *User = Use.getUser();
7951 if (!User->hasOneUse() || !User->isMachineOpcode() ||
7952 User->getMachineOpcode() != TargetOpcode::EXTRACT_SUBREG)
7953 return false;
7954
7955 switch (User->getConstantOperandVal(1)) {
7956 case SystemZ::subreg_l64:
7957 if (LoPart)
7958 return false;
7959 LoPart = User;
7960 break;
7961 case SystemZ::subreg_h64:
7962 if (HiPart)
7963 return false;
7964 HiPart = User;
7965 break;
7966 default:
7967 return false;
7968 }
7969 }
7970 return true;
7971}
7972
7973SDValue SystemZTargetLowering::combineLOAD(
7974 SDNode *N, DAGCombinerInfo &DCI) const {
7975 SelectionDAG &DAG = DCI.DAG;
7976 EVT LdVT = N->getValueType(0);
7977 if (auto *LN = dyn_cast<LoadSDNode>(N)) {
7978 if (LN->getAddressSpace() == SYSTEMZAS::PTR32) {
7979 MVT PtrVT = getPointerTy(DAG.getDataLayout());
7980 MVT LoadNodeVT = LN->getBasePtr().getSimpleValueType();
7981 if (PtrVT != LoadNodeVT) {
7982 SDLoc DL(LN);
7983 SDValue AddrSpaceCast = DAG.getAddrSpaceCast(
7984 DL, PtrVT, LN->getBasePtr(), SYSTEMZAS::PTR32, 0);
7985 return DAG.getExtLoad(LN->getExtensionType(), DL, LN->getValueType(0),
7986 LN->getChain(), AddrSpaceCast, LN->getMemoryVT(),
7987 LN->getMemOperand());
7988 }
7989 }
7990 }
7991 SDLoc DL(N);
7992
7993 // Replace a 128-bit load that is used solely to move its value into GPRs
7994 // by separate loads of both halves.
7995 LoadSDNode *LD = cast<LoadSDNode>(N);
7996 if (LD->isSimple() && ISD::isNormalLoad(LD)) {
7997 SDNode *LoPart, *HiPart;
7998 if ((LdVT == MVT::i128 && isI128MovedToParts(LD, LoPart, HiPart)) ||
7999 (LdVT == MVT::f128 && isF128MovedToParts(LD, LoPart, HiPart))) {
8000 // Rewrite each extraction as an independent load.
8001 SmallVector<SDValue, 2> ArgChains;
8002 if (HiPart) {
8003 SDValue EltLoad = DAG.getLoad(
8004 HiPart->getValueType(0), DL, LD->getChain(), LD->getBasePtr(),
8005 LD->getPointerInfo(), LD->getBaseAlign(),
8006 LD->getMemOperand()->getFlags(), LD->getAAInfo());
8007
8008 DCI.CombineTo(HiPart, EltLoad, true);
8009 ArgChains.push_back(EltLoad.getValue(1));
8010 }
8011 if (LoPart) {
8012 SDValue EltLoad = DAG.getLoad(
8013 LoPart->getValueType(0), DL, LD->getChain(),
8014 DAG.getObjectPtrOffset(DL, LD->getBasePtr(), TypeSize::getFixed(8)),
8015 LD->getPointerInfo().getWithOffset(8), LD->getBaseAlign(),
8016 LD->getMemOperand()->getFlags(), LD->getAAInfo());
8017
8018 DCI.CombineTo(LoPart, EltLoad, true);
8019 ArgChains.push_back(EltLoad.getValue(1));
8020 }
8021
8022 // Collect all chains via TokenFactor.
8023 SDValue Chain = DAG.getNode(ISD::TokenFactor, DL, MVT::Other, ArgChains);
8024 DAG.ReplaceAllUsesOfValueWith(SDValue(N, 1), Chain);
8025 DCI.AddToWorklist(Chain.getNode());
8026 return SDValue(N, 0);
8027 }
8028 }
8029
8030 if (LdVT.isVector() || LdVT.isInteger())
8031 return SDValue();
8032 // Transform a scalar load that is REPLICATEd as well as having other
8033 // use(s) to the form where the other use(s) use the first element of the
8034 // REPLICATE instead of the load. Otherwise instruction selection will not
8035 // produce a VLREP. Avoid extracting to a GPR, so only do this for floating
8036 // point loads.
8037
8038 SDValue Replicate;
8039 SmallVector<SDNode*, 8> OtherUses;
8040 for (SDUse &Use : N->uses()) {
8041 if (Use.getUser()->getOpcode() == SystemZISD::REPLICATE) {
8042 if (Replicate)
8043 return SDValue(); // Should never happen
8044 Replicate = SDValue(Use.getUser(), 0);
8045 } else if (Use.getResNo() == 0)
8046 OtherUses.push_back(Use.getUser());
8047 }
8048 if (!Replicate || OtherUses.empty())
8049 return SDValue();
8050
8051 SDValue Extract0 = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, DL, LdVT,
8052 Replicate, DAG.getConstant(0, DL, MVT::i32));
8053 // Update uses of the loaded Value while preserving old chains.
8054 for (SDNode *U : OtherUses) {
8056 for (SDValue Op : U->ops())
8057 Ops.push_back((Op.getNode() == N && Op.getResNo() == 0) ? Extract0 : Op);
8058 DAG.UpdateNodeOperands(U, Ops);
8059 }
8060 return SDValue(N, 0);
8061}
8062
8063bool SystemZTargetLowering::canLoadStoreByteSwapped(EVT VT) const {
8064 if (VT == MVT::i16 || VT == MVT::i32 || VT == MVT::i64)
8065 return true;
8066 if (Subtarget.hasVectorEnhancements2())
8067 if (VT == MVT::v8i16 || VT == MVT::v4i32 || VT == MVT::v2i64 || VT == MVT::i128)
8068 return true;
8069 return false;
8070}
8071
8073 if (!VT.isVector() || !VT.isSimple() ||
8074 VT.getSizeInBits() != 128 ||
8075 VT.getScalarSizeInBits() % 8 != 0)
8076 return false;
8077
8078 unsigned NumElts = VT.getVectorNumElements();
8079 for (unsigned i = 0; i < NumElts; ++i) {
8080 if (M[i] < 0) continue; // ignore UNDEF indices
8081 if ((unsigned) M[i] != NumElts - 1 - i)
8082 return false;
8083 }
8084
8085 return true;
8086}
8087
8088static bool isOnlyUsedByStores(SDValue StoredVal, SelectionDAG &DAG) {
8089 for (auto *U : StoredVal->users()) {
8090 if (StoreSDNode *ST = dyn_cast<StoreSDNode>(U)) {
8091 EVT CurrMemVT = ST->getMemoryVT().getScalarType();
8092 if (CurrMemVT.isRound() && CurrMemVT.getStoreSize() <= 16)
8093 continue;
8094 } else if (isa<BuildVectorSDNode>(U)) {
8095 SDValue BuildVector = SDValue(U, 0);
8096 if (DAG.isSplatValue(BuildVector, true/*AllowUndefs*/) &&
8097 isOnlyUsedByStores(BuildVector, DAG))
8098 continue;
8099 }
8100 return false;
8101 }
8102 return true;
8103}
8104
8105static bool isI128MovedFromParts(SDValue Val, SDValue &LoPart,
8106 SDValue &HiPart) {
8107 if (Val.getOpcode() != ISD::OR || !Val.getNode()->hasOneUse())
8108 return false;
8109
8110 SDValue Op0 = Val.getOperand(0);
8111 SDValue Op1 = Val.getOperand(1);
8112
8113 if (Op0.getOpcode() == ISD::SHL)
8114 std::swap(Op0, Op1);
8115 if (Op1.getOpcode() != ISD::SHL || !Op1.getNode()->hasOneUse() ||
8116 Op1.getOperand(1).getOpcode() != ISD::Constant ||
8117 Op1.getConstantOperandVal(1) != 64)
8118 return false;
8119 Op1 = Op1.getOperand(0);
8120
8121 if (Op0.getOpcode() != ISD::ZERO_EXTEND || !Op0.getNode()->hasOneUse() ||
8122 Op0.getOperand(0).getValueType() != MVT::i64)
8123 return false;
8124 if (Op1.getOpcode() != ISD::ANY_EXTEND || !Op1.getNode()->hasOneUse() ||
8125 Op1.getOperand(0).getValueType() != MVT::i64)
8126 return false;
8127
8128 LoPart = Op0.getOperand(0);
8129 HiPart = Op1.getOperand(0);
8130 return true;
8131}
8132
8133static bool isF128MovedFromParts(SDValue Val, SDValue &LoPart,
8134 SDValue &HiPart) {
8135 if (!Val.getNode()->hasOneUse() || !Val.isMachineOpcode() ||
8136 Val.getMachineOpcode() != TargetOpcode::REG_SEQUENCE)
8137 return false;
8138
8139 if (Val->getNumOperands() != 5 ||
8140 Val->getOperand(0)->getAsZExtVal() != SystemZ::FP128BitRegClassID ||
8141 Val->getOperand(2)->getAsZExtVal() != SystemZ::subreg_l64 ||
8142 Val->getOperand(4)->getAsZExtVal() != SystemZ::subreg_h64)
8143 return false;
8144
8145 LoPart = Val->getOperand(1);
8146 HiPart = Val->getOperand(3);
8147 return true;
8148}
8149
8150SDValue SystemZTargetLowering::combineSTORE(
8151 SDNode *N, DAGCombinerInfo &DCI) const {
8152 SelectionDAG &DAG = DCI.DAG;
8153 auto *SN = cast<StoreSDNode>(N);
8154 auto &Op1 = N->getOperand(1);
8155 EVT MemVT = SN->getMemoryVT();
8156
8157 if (SN->getAddressSpace() == SYSTEMZAS::PTR32) {
8158 MVT PtrVT = getPointerTy(DAG.getDataLayout());
8159 MVT StoreNodeVT = SN->getBasePtr().getSimpleValueType();
8160 if (PtrVT != StoreNodeVT) {
8161 SDLoc DL(SN);
8162 SDValue AddrSpaceCast = DAG.getAddrSpaceCast(DL, PtrVT, SN->getBasePtr(),
8163 SYSTEMZAS::PTR32, 0);
8164 return DAG.getStore(SN->getChain(), DL, SN->getValue(), AddrSpaceCast,
8165 SN->getPointerInfo(), SN->getBaseAlign(),
8166 SN->getMemOperand()->getFlags(), SN->getAAInfo());
8167 }
8168 }
8169
8170 // If we have (truncstoreiN (extract_vector_elt X, Y), Z) then it is better
8171 // for the extraction to be done on a vMiN value, so that we can use VSTE.
8172 // If X has wider elements then convert it to:
8173 // (truncstoreiN (extract_vector_elt (bitcast X), Y2), Z).
8174 if (MemVT.isInteger() && SN->isTruncatingStore()) {
8175 if (SDValue Value =
8176 combineTruncateExtract(SDLoc(N), MemVT, SN->getValue(), DCI)) {
8177 DCI.AddToWorklist(Value.getNode());
8178
8179 // Rewrite the store with the new form of stored value.
8180 return DAG.getTruncStore(SN->getChain(), SDLoc(SN), Value,
8181 SN->getBasePtr(), SN->getMemoryVT(),
8182 SN->getMemOperand());
8183 }
8184 }
8185
8186 // combine STORE (LOAD_STACK_GUARD) into MOV_STACKGUARD_DAG
8187 if (Op1->isMachineOpcode() &&
8188 (Op1->getMachineOpcode() == SystemZ::LOAD_STACK_GUARD)) {
8189 // Obtain the frame index the store was targeting.
8190 int FI = cast<FrameIndexSDNode>(SN->getOperand(2))->getIndex();
8191 // Prepare operands of the MOV_STACKGUARD ISD Node - Chain and FrameIndex.
8192 SDValue Ops[] = {SN->getChain(), DAG.getTargetFrameIndex(FI, MVT::i64)};
8193 return DAG.getNode(SystemZISD::MOV_STACKGUARD, SDLoc(SN), MVT::Other, Ops);
8194 }
8195
8196 // Combine STORE (BSWAP) into STRVH/STRV/STRVG/VSTBR
8197 if (!SN->isTruncatingStore() &&
8198 Op1.getOpcode() == ISD::BSWAP &&
8199 Op1.getNode()->hasOneUse() &&
8200 canLoadStoreByteSwapped(Op1.getValueType())) {
8201
8202 SDValue BSwapOp = Op1.getOperand(0);
8203
8204 if (BSwapOp.getValueType() == MVT::i16)
8205 BSwapOp = DAG.getNode(ISD::ANY_EXTEND, SDLoc(N), MVT::i32, BSwapOp);
8206
8207 SDValue Ops[] = {
8208 N->getOperand(0), BSwapOp, N->getOperand(2)
8209 };
8210
8211 return
8212 DAG.getMemIntrinsicNode(SystemZISD::STRV, SDLoc(N), DAG.getVTList(MVT::Other),
8213 Ops, MemVT, SN->getMemOperand());
8214 }
8215 // Combine STORE (element-swap) into VSTER
8216 if (!SN->isTruncatingStore() &&
8217 Op1.getOpcode() == ISD::VECTOR_SHUFFLE &&
8218 Op1.getNode()->hasOneUse() &&
8219 Subtarget.hasVectorEnhancements2()) {
8220 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(Op1.getNode());
8221 ArrayRef<int> ShuffleMask = SVN->getMask();
8222 if (isVectorElementSwap(ShuffleMask, Op1.getValueType())) {
8223 SDValue Ops[] = {
8224 N->getOperand(0), Op1.getOperand(0), N->getOperand(2)
8225 };
8226
8227 return DAG.getMemIntrinsicNode(SystemZISD::VSTER, SDLoc(N),
8228 DAG.getVTList(MVT::Other),
8229 Ops, MemVT, SN->getMemOperand());
8230 }
8231 }
8232
8233 // Combine STORE (READCYCLECOUNTER) into STCKF.
8234 if (!SN->isTruncatingStore() &&
8236 Op1.hasOneUse() &&
8237 N->getOperand(0).reachesChainWithoutSideEffects(SDValue(Op1.getNode(), 1))) {
8238 SDValue Ops[] = { Op1.getOperand(0), N->getOperand(2) };
8239 return DAG.getMemIntrinsicNode(SystemZISD::STCKF, SDLoc(N),
8240 DAG.getVTList(MVT::Other),
8241 Ops, MemVT, SN->getMemOperand());
8242 }
8243
8244 // Transform a store of a 128-bit value moved from parts into two stores.
8245 if (SN->isSimple() && ISD::isNormalStore(SN)) {
8246 SDValue LoPart, HiPart;
8247 if ((MemVT == MVT::i128 && isI128MovedFromParts(Op1, LoPart, HiPart)) ||
8248 (MemVT == MVT::f128 && isF128MovedFromParts(Op1, LoPart, HiPart))) {
8249 SDLoc DL(SN);
8250 SDValue Chain0 = DAG.getStore(
8251 SN->getChain(), DL, HiPart, SN->getBasePtr(), SN->getPointerInfo(),
8252 SN->getBaseAlign(), SN->getMemOperand()->getFlags(), SN->getAAInfo());
8253 SDValue Chain1 = DAG.getStore(
8254 SN->getChain(), DL, LoPart,
8255 DAG.getObjectPtrOffset(DL, SN->getBasePtr(), TypeSize::getFixed(8)),
8256 SN->getPointerInfo().getWithOffset(8), SN->getBaseAlign(),
8257 SN->getMemOperand()->getFlags(), SN->getAAInfo());
8258
8259 return DAG.getNode(ISD::TokenFactor, DL, MVT::Other, Chain0, Chain1);
8260 }
8261 }
8262
8263 // Replicate a reg or immediate with VREP instead of scalar multiply or
8264 // immediate load. It seems best to do this during the first DAGCombine as
8265 // it is straight-forward to handle the zero-extend node in the initial
8266 // DAG, and also not worry about the keeping the new MemVT legal (e.g. when
8267 // extracting an i16 element from a v16i8 vector).
8268 if (Subtarget.hasVector() && DCI.Level == BeforeLegalizeTypes &&
8269 isOnlyUsedByStores(Op1, DAG)) {
8270 SDValue Word = SDValue();
8271 EVT WordVT;
8272
8273 // Find a replicated immediate and return it if found in Word and its
8274 // type in WordVT.
8275 auto FindReplicatedImm = [&](ConstantSDNode *C, unsigned TotBytes) {
8276 // Some constants are better handled with a scalar store.
8277 if (C->getAPIntValue().getBitWidth() > 64 || C->isAllOnes() ||
8278 isInt<16>(C->getSExtValue()) || MemVT.getStoreSize() <= 2)
8279 return;
8280
8281 APInt Val = C->getAPIntValue();
8282 // Truncate Val in case of a truncating store.
8283 if (!llvm::isUIntN(TotBytes * 8, Val.getZExtValue())) {
8284 assert(SN->isTruncatingStore() &&
8285 "Non-truncating store and immediate value does not fit?");
8286 Val = Val.trunc(TotBytes * 8);
8287 }
8288
8289 SystemZVectorConstantInfo VCI(APInt(TotBytes * 8, Val.getZExtValue()));
8290 if (VCI.isVectorConstantLegal(Subtarget) &&
8291 VCI.Opcode == SystemZISD::REPLICATE) {
8292 Word = DAG.getConstant(VCI.OpVals[0], SDLoc(SN), MVT::i32);
8293 WordVT = VCI.VecVT.getScalarType();
8294 }
8295 };
8296
8297 // Find a replicated register and return it if found in Word and its type
8298 // in WordVT.
8299 auto FindReplicatedReg = [&](SDValue MulOp) {
8300 EVT MulVT = MulOp.getValueType();
8301 if (MulOp->getOpcode() == ISD::MUL &&
8302 (MulVT == MVT::i16 || MulVT == MVT::i32 || MulVT == MVT::i64)) {
8303 // Find a zero extended value and its type.
8304 SDValue LHS = MulOp->getOperand(0);
8305 if (LHS->getOpcode() == ISD::ZERO_EXTEND)
8306 WordVT = LHS->getOperand(0).getValueType();
8307 else if (LHS->getOpcode() == ISD::AssertZext)
8308 WordVT = cast<VTSDNode>(LHS->getOperand(1))->getVT();
8309 else
8310 return;
8311 // Find a replicating constant, e.g. 0x00010001.
8312 if (auto *C = dyn_cast<ConstantSDNode>(MulOp->getOperand(1))) {
8313 SystemZVectorConstantInfo VCI(
8314 APInt(MulVT.getSizeInBits(), C->getZExtValue()));
8315 if (VCI.isVectorConstantLegal(Subtarget) &&
8316 VCI.Opcode == SystemZISD::REPLICATE && VCI.OpVals[0] == 1 &&
8317 WordVT == VCI.VecVT.getScalarType())
8318 Word = DAG.getZExtOrTrunc(LHS->getOperand(0), SDLoc(SN), WordVT);
8319 }
8320 }
8321 };
8322
8323 if (isa<BuildVectorSDNode>(Op1) &&
8324 DAG.isSplatValue(Op1, true/*AllowUndefs*/)) {
8325 SDValue SplatVal = Op1->getOperand(0);
8326 if (auto *C = dyn_cast<ConstantSDNode>(SplatVal))
8327 FindReplicatedImm(C, SplatVal.getValueType().getStoreSize());
8328 else
8329 FindReplicatedReg(SplatVal);
8330 } else {
8331 if (auto *C = dyn_cast<ConstantSDNode>(Op1))
8332 FindReplicatedImm(C, MemVT.getStoreSize());
8333 else
8334 FindReplicatedReg(Op1);
8335 }
8336
8337 if (Word != SDValue()) {
8338 assert(MemVT.getSizeInBits() % WordVT.getSizeInBits() == 0 &&
8339 "Bad type handling");
8340 unsigned NumElts = MemVT.getSizeInBits() / WordVT.getSizeInBits();
8341 EVT SplatVT = EVT::getVectorVT(*DAG.getContext(), WordVT, NumElts);
8342 SDValue SplatVal = DAG.getSplatVector(SplatVT, SDLoc(SN), Word);
8343 return DAG.getStore(SN->getChain(), SDLoc(SN), SplatVal,
8344 SN->getBasePtr(), SN->getMemOperand());
8345 }
8346 }
8347
8348 return SDValue();
8349}
8350
8351SDValue SystemZTargetLowering::combineVECTOR_SHUFFLE(
8352 SDNode *N, DAGCombinerInfo &DCI) const {
8353 SelectionDAG &DAG = DCI.DAG;
8354 // Combine element-swap (LOAD) into VLER
8355 if (ISD::isNON_EXTLoad(N->getOperand(0).getNode()) &&
8356 N->getOperand(0).hasOneUse() &&
8357 Subtarget.hasVectorEnhancements2()) {
8358 ShuffleVectorSDNode *SVN = cast<ShuffleVectorSDNode>(N);
8359 ArrayRef<int> ShuffleMask = SVN->getMask();
8360 if (isVectorElementSwap(ShuffleMask, N->getValueType(0))) {
8361 SDValue Load = N->getOperand(0);
8362 LoadSDNode *LD = cast<LoadSDNode>(Load);
8363
8364 // Create the element-swapping load.
8365 SDValue Ops[] = {
8366 LD->getChain(), // Chain
8367 LD->getBasePtr() // Ptr
8368 };
8369 SDValue ESLoad =
8370 DAG.getMemIntrinsicNode(SystemZISD::VLER, SDLoc(N),
8371 DAG.getVTList(LD->getValueType(0), MVT::Other),
8372 Ops, LD->getMemoryVT(), LD->getMemOperand());
8373
8374 // First, combine the VECTOR_SHUFFLE away. This makes the value produced
8375 // by the load dead.
8376 DCI.CombineTo(N, ESLoad);
8377
8378 // Next, combine the load away, we give it a bogus result value but a real
8379 // chain result. The result value is dead because the shuffle is dead.
8380 DCI.CombineTo(Load.getNode(), ESLoad, ESLoad.getValue(1));
8381
8382 // Return N so it doesn't get rechecked!
8383 return SDValue(N, 0);
8384 }
8385 }
8386
8387 return SDValue();
8388}
8389
8390SDValue SystemZTargetLowering::combineEXTRACT_VECTOR_ELT(
8391 SDNode *N, DAGCombinerInfo &DCI) const {
8392 SelectionDAG &DAG = DCI.DAG;
8393
8394 if (!Subtarget.hasVector())
8395 return SDValue();
8396
8397 // Look through bitcasts that retain the number of vector elements.
8398 SDValue Op = N->getOperand(0);
8399 if (Op.getOpcode() == ISD::BITCAST &&
8400 Op.getValueType().isVector() &&
8401 Op.getOperand(0).getValueType().isVector() &&
8402 Op.getValueType().getVectorNumElements() ==
8403 Op.getOperand(0).getValueType().getVectorNumElements())
8404 Op = Op.getOperand(0);
8405
8406 // Pull BSWAP out of a vector extraction.
8407 if (Op.getOpcode() == ISD::BSWAP && Op.hasOneUse()) {
8408 EVT VecVT = Op.getValueType();
8409 EVT EltVT = VecVT.getVectorElementType();
8410 Op = DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(N), EltVT,
8411 Op.getOperand(0), N->getOperand(1));
8412 DCI.AddToWorklist(Op.getNode());
8413 Op = DAG.getNode(ISD::BSWAP, SDLoc(N), EltVT, Op);
8414 if (EltVT != N->getValueType(0)) {
8415 DCI.AddToWorklist(Op.getNode());
8416 Op = DAG.getNode(ISD::BITCAST, SDLoc(N), N->getValueType(0), Op);
8417 }
8418 return Op;
8419 }
8420
8421 // Try to simplify a vector extraction.
8422 if (auto *IndexN = dyn_cast<ConstantSDNode>(N->getOperand(1))) {
8423 SDValue Op0 = N->getOperand(0);
8424 EVT VecVT = Op0.getValueType();
8425 if (canTreatAsByteVector(VecVT))
8426 return combineExtract(SDLoc(N), N->getValueType(0), VecVT, Op0,
8427 IndexN->getZExtValue(), DCI, false);
8428 }
8429 return SDValue();
8430}
8431
8432SDValue SystemZTargetLowering::combineJOIN_DWORDS(
8433 SDNode *N, DAGCombinerInfo &DCI) const {
8434 SelectionDAG &DAG = DCI.DAG;
8435 // (join_dwords X, X) == (replicate X)
8436 if (N->getOperand(0) == N->getOperand(1))
8437 return DAG.getNode(SystemZISD::REPLICATE, SDLoc(N), N->getValueType(0),
8438 N->getOperand(0));
8439 return SDValue();
8440}
8441
8443 SDValue Chain1 = N1->getOperand(0);
8444 SDValue Chain2 = N2->getOperand(0);
8445
8446 // Trivial case: both nodes take the same chain.
8447 if (Chain1 == Chain2)
8448 return Chain1;
8449
8450 // FIXME - we could handle more complex cases via TokenFactor,
8451 // assuming we can verify that this would not create a cycle.
8452 return SDValue();
8453}
8454
8455SDValue SystemZTargetLowering::combineFP_ROUND(
8456 SDNode *N, DAGCombinerInfo &DCI) const {
8457
8458 if (!Subtarget.hasVector())
8459 return SDValue();
8460
8461 // (fpround (extract_vector_elt X 0))
8462 // (fpround (extract_vector_elt X 1)) ->
8463 // (extract_vector_elt (VROUND X) 0)
8464 // (extract_vector_elt (VROUND X) 2)
8465 //
8466 // This is a special case since the target doesn't really support v2f32s.
8467 unsigned OpNo = N->isStrictFPOpcode() ? 1 : 0;
8468 SelectionDAG &DAG = DCI.DAG;
8469 SDValue Op0 = N->getOperand(OpNo);
8470 if (N->getValueType(0) == MVT::f32 && Op0.hasOneUse() &&
8472 Op0.getOperand(0).getValueType() == MVT::v2f64 &&
8473 Op0.getOperand(1).getOpcode() == ISD::Constant &&
8474 Op0.getConstantOperandVal(1) == 0) {
8475 SDValue Vec = Op0.getOperand(0);
8476 for (auto *U : Vec->users()) {
8477 if (U != Op0.getNode() && U->hasOneUse() &&
8478 U->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
8479 U->getOperand(0) == Vec &&
8480 U->getOperand(1).getOpcode() == ISD::Constant &&
8481 U->getConstantOperandVal(1) == 1) {
8482 SDValue OtherRound = SDValue(*U->user_begin(), 0);
8483 if (OtherRound.getOpcode() == N->getOpcode() &&
8484 OtherRound.getOperand(OpNo) == SDValue(U, 0) &&
8485 OtherRound.getValueType() == MVT::f32) {
8486 SDValue VRound, Chain;
8487 if (N->isStrictFPOpcode()) {
8488 Chain = MergeInputChains(N, OtherRound.getNode());
8489 if (!Chain)
8490 continue;
8491 VRound = DAG.getNode(SystemZISD::STRICT_VROUND, SDLoc(N),
8492 {MVT::v4f32, MVT::Other}, {Chain, Vec});
8493 Chain = VRound.getValue(1);
8494 } else
8495 VRound = DAG.getNode(SystemZISD::VROUND, SDLoc(N),
8496 MVT::v4f32, Vec);
8497 DCI.AddToWorklist(VRound.getNode());
8498 SDValue Extract1 =
8499 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(U), MVT::f32,
8500 VRound, DAG.getConstant(2, SDLoc(U), MVT::i32));
8501 DCI.AddToWorklist(Extract1.getNode());
8502 DAG.ReplaceAllUsesOfValueWith(OtherRound, Extract1);
8503 if (Chain)
8504 DAG.ReplaceAllUsesOfValueWith(OtherRound.getValue(1), Chain);
8505 SDValue Extract0 =
8506 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(Op0), MVT::f32,
8507 VRound, DAG.getConstant(0, SDLoc(Op0), MVT::i32));
8508 if (Chain)
8509 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op0),
8510 N->getVTList(), Extract0, Chain);
8511 return Extract0;
8512 }
8513 }
8514 }
8515 }
8516 return SDValue();
8517}
8518
8519SDValue SystemZTargetLowering::combineFP_EXTEND(
8520 SDNode *N, DAGCombinerInfo &DCI) const {
8521
8522 if (!Subtarget.hasVector())
8523 return SDValue();
8524
8525 // (fpextend (extract_vector_elt X 0))
8526 // (fpextend (extract_vector_elt X 2)) ->
8527 // (extract_vector_elt (VEXTEND X) 0)
8528 // (extract_vector_elt (VEXTEND X) 1)
8529 //
8530 // This is a special case since the target doesn't really support v2f32s.
8531 unsigned OpNo = N->isStrictFPOpcode() ? 1 : 0;
8532 SelectionDAG &DAG = DCI.DAG;
8533 SDValue Op0 = N->getOperand(OpNo);
8534 if (N->getValueType(0) == MVT::f64 && Op0.hasOneUse() &&
8536 Op0.getOperand(0).getValueType() == MVT::v4f32 &&
8537 Op0.getOperand(1).getOpcode() == ISD::Constant &&
8538 Op0.getConstantOperandVal(1) == 0) {
8539 SDValue Vec = Op0.getOperand(0);
8540 for (auto *U : Vec->users()) {
8541 if (U != Op0.getNode() && U->hasOneUse() &&
8542 U->getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
8543 U->getOperand(0) == Vec &&
8544 U->getOperand(1).getOpcode() == ISD::Constant &&
8545 U->getConstantOperandVal(1) == 2) {
8546 SDValue OtherExtend = SDValue(*U->user_begin(), 0);
8547 if (OtherExtend.getOpcode() == N->getOpcode() &&
8548 OtherExtend.getOperand(OpNo) == SDValue(U, 0) &&
8549 OtherExtend.getValueType() == MVT::f64) {
8550 SDValue VExtend, Chain;
8551 if (N->isStrictFPOpcode()) {
8552 Chain = MergeInputChains(N, OtherExtend.getNode());
8553 if (!Chain)
8554 continue;
8555 VExtend = DAG.getNode(SystemZISD::STRICT_VEXTEND, SDLoc(N),
8556 {MVT::v2f64, MVT::Other}, {Chain, Vec});
8557 Chain = VExtend.getValue(1);
8558 } else
8559 VExtend = DAG.getNode(SystemZISD::VEXTEND, SDLoc(N),
8560 MVT::v2f64, Vec);
8561 DCI.AddToWorklist(VExtend.getNode());
8562 SDValue Extract1 =
8563 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(U), MVT::f64,
8564 VExtend, DAG.getConstant(1, SDLoc(U), MVT::i32));
8565 DCI.AddToWorklist(Extract1.getNode());
8566 DAG.ReplaceAllUsesOfValueWith(OtherExtend, Extract1);
8567 if (Chain)
8568 DAG.ReplaceAllUsesOfValueWith(OtherExtend.getValue(1), Chain);
8569 SDValue Extract0 =
8570 DAG.getNode(ISD::EXTRACT_VECTOR_ELT, SDLoc(Op0), MVT::f64,
8571 VExtend, DAG.getConstant(0, SDLoc(Op0), MVT::i32));
8572 if (Chain)
8573 return DAG.getNode(ISD::MERGE_VALUES, SDLoc(Op0),
8574 N->getVTList(), Extract0, Chain);
8575 return Extract0;
8576 }
8577 }
8578 }
8579 }
8580 return SDValue();
8581}
8582
8583SDValue SystemZTargetLowering::combineINT_TO_FP(
8584 SDNode *N, DAGCombinerInfo &DCI) const {
8585 if (DCI.Level != BeforeLegalizeTypes)
8586 return SDValue();
8587 SelectionDAG &DAG = DCI.DAG;
8588 LLVMContext &Ctx = *DAG.getContext();
8589 unsigned Opcode = N->getOpcode();
8590 EVT OutVT = N->getValueType(0);
8591 Type *OutLLVMTy = OutVT.getTypeForEVT(Ctx);
8592 SDValue Op = N->getOperand(0);
8593 unsigned OutScalarBits = OutLLVMTy->getScalarSizeInBits();
8594 unsigned InScalarBits = Op->getValueType(0).getScalarSizeInBits();
8595
8596 // Insert an extension before type-legalization to avoid scalarization, e.g.:
8597 // v2f64 = uint_to_fp v2i16
8598 // =>
8599 // v2f64 = uint_to_fp (v2i64 zero_extend v2i16)
8600 if (OutLLVMTy->isVectorTy() && OutScalarBits > InScalarBits &&
8601 OutScalarBits <= 64) {
8602 unsigned NumElts = cast<FixedVectorType>(OutLLVMTy)->getNumElements();
8603 EVT ExtVT = EVT::getVectorVT(
8604 Ctx, EVT::getIntegerVT(Ctx, OutLLVMTy->getScalarSizeInBits()), NumElts);
8605 unsigned ExtOpcode =
8607 SDValue ExtOp = DAG.getNode(ExtOpcode, SDLoc(N), ExtVT, Op);
8608 return DAG.getNode(Opcode, SDLoc(N), OutVT, ExtOp);
8609 }
8610 return SDValue();
8611}
8612
8613SDValue SystemZTargetLowering::combineFCOPYSIGN(
8614 SDNode *N, DAGCombinerInfo &DCI) const {
8615 SelectionDAG &DAG = DCI.DAG;
8616 EVT VT = N->getValueType(0);
8617 SDValue ValOp = N->getOperand(0);
8618 SDValue SignOp = N->getOperand(1);
8619
8620 // Remove the rounding which is not needed.
8621 if (SignOp.getOpcode() == ISD::FP_ROUND) {
8622 SDValue WideOp = SignOp.getOperand(0);
8623 return DAG.getNode(ISD::FCOPYSIGN, SDLoc(N), VT, ValOp, WideOp);
8624 }
8625
8626 return SDValue();
8627}
8628
8629SDValue SystemZTargetLowering::combineBSWAP(
8630 SDNode *N, DAGCombinerInfo &DCI) const {
8631 SelectionDAG &DAG = DCI.DAG;
8632 // Combine BSWAP (LOAD) into LRVH/LRV/LRVG/VLBR
8633 if (ISD::isNON_EXTLoad(N->getOperand(0).getNode()) &&
8634 N->getOperand(0).hasOneUse() &&
8635 canLoadStoreByteSwapped(N->getValueType(0))) {
8636 SDValue Load = N->getOperand(0);
8637 LoadSDNode *LD = cast<LoadSDNode>(Load);
8638
8639 // Create the byte-swapping load.
8640 SDValue Ops[] = {
8641 LD->getChain(), // Chain
8642 LD->getBasePtr() // Ptr
8643 };
8644 EVT LoadVT = N->getValueType(0);
8645 if (LoadVT == MVT::i16)
8646 LoadVT = MVT::i32;
8647 SDValue BSLoad =
8648 DAG.getMemIntrinsicNode(SystemZISD::LRV, SDLoc(N),
8649 DAG.getVTList(LoadVT, MVT::Other),
8650 Ops, LD->getMemoryVT(), LD->getMemOperand());
8651
8652 // If this is an i16 load, insert the truncate.
8653 SDValue ResVal = BSLoad;
8654 if (N->getValueType(0) == MVT::i16)
8655 ResVal = DAG.getNode(ISD::TRUNCATE, SDLoc(N), MVT::i16, BSLoad);
8656
8657 // First, combine the bswap away. This makes the value produced by the
8658 // load dead.
8659 DCI.CombineTo(N, ResVal);
8660
8661 // Next, combine the load away, we give it a bogus result value but a real
8662 // chain result. The result value is dead because the bswap is dead.
8663 DCI.CombineTo(Load.getNode(), ResVal, BSLoad.getValue(1));
8664
8665 // Return N so it doesn't get rechecked!
8666 return SDValue(N, 0);
8667 }
8668
8669 // Look through bitcasts that retain the number of vector elements.
8670 SDValue Op = N->getOperand(0);
8671 if (Op.getOpcode() == ISD::BITCAST &&
8672 Op.getValueType().isVector() &&
8673 Op.getOperand(0).getValueType().isVector() &&
8674 Op.getValueType().getVectorNumElements() ==
8675 Op.getOperand(0).getValueType().getVectorNumElements())
8676 Op = Op.getOperand(0);
8677
8678 // Push BSWAP into a vector insertion if at least one side then simplifies.
8679 if (Op.getOpcode() == ISD::INSERT_VECTOR_ELT && Op.hasOneUse()) {
8680 SDValue Vec = Op.getOperand(0);
8681 SDValue Elt = Op.getOperand(1);
8682 SDValue Idx = Op.getOperand(2);
8683
8685 Vec.getOpcode() == ISD::BSWAP || Vec.isUndef() ||
8687 Elt.getOpcode() == ISD::BSWAP || Elt.isUndef() ||
8688 (canLoadStoreByteSwapped(N->getValueType(0)) &&
8689 ISD::isNON_EXTLoad(Elt.getNode()) && Elt.hasOneUse())) {
8690 EVT VecVT = N->getValueType(0);
8691 EVT EltVT = N->getValueType(0).getVectorElementType();
8692 if (VecVT != Vec.getValueType()) {
8693 Vec = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Vec);
8694 DCI.AddToWorklist(Vec.getNode());
8695 }
8696 if (EltVT != Elt.getValueType()) {
8697 Elt = DAG.getNode(ISD::BITCAST, SDLoc(N), EltVT, Elt);
8698 DCI.AddToWorklist(Elt.getNode());
8699 }
8700 Vec = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Vec);
8701 DCI.AddToWorklist(Vec.getNode());
8702 Elt = DAG.getNode(ISD::BSWAP, SDLoc(N), EltVT, Elt);
8703 DCI.AddToWorklist(Elt.getNode());
8704 return DAG.getNode(ISD::INSERT_VECTOR_ELT, SDLoc(N), VecVT,
8705 Vec, Elt, Idx);
8706 }
8707 }
8708
8709 // Push BSWAP into a vector shuffle if at least one side then simplifies.
8710 ShuffleVectorSDNode *SV = dyn_cast<ShuffleVectorSDNode>(Op);
8711 if (SV && Op.hasOneUse()) {
8712 SDValue Op0 = Op.getOperand(0);
8713 SDValue Op1 = Op.getOperand(1);
8714
8716 Op0.getOpcode() == ISD::BSWAP || Op0.isUndef() ||
8718 Op1.getOpcode() == ISD::BSWAP || Op1.isUndef()) {
8719 EVT VecVT = N->getValueType(0);
8720 if (VecVT != Op0.getValueType()) {
8721 Op0 = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Op0);
8722 DCI.AddToWorklist(Op0.getNode());
8723 }
8724 if (VecVT != Op1.getValueType()) {
8725 Op1 = DAG.getNode(ISD::BITCAST, SDLoc(N), VecVT, Op1);
8726 DCI.AddToWorklist(Op1.getNode());
8727 }
8728 Op0 = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Op0);
8729 DCI.AddToWorklist(Op0.getNode());
8730 Op1 = DAG.getNode(ISD::BSWAP, SDLoc(N), VecVT, Op1);
8731 DCI.AddToWorklist(Op1.getNode());
8732 return DAG.getVectorShuffle(VecVT, SDLoc(N), Op0, Op1, SV->getMask());
8733 }
8734 }
8735
8736 return SDValue();
8737}
8738
8739SDValue SystemZTargetLowering::combineSETCC(
8740 SDNode *N, DAGCombinerInfo &DCI) const {
8741 SelectionDAG &DAG = DCI.DAG;
8742 const ISD::CondCode CC = cast<CondCodeSDNode>(N->getOperand(2))->get();
8743 const SDValue LHS = N->getOperand(0);
8744 const SDValue RHS = N->getOperand(1);
8745 bool CmpNull = isNullConstant(RHS);
8746 bool CmpAllOnes = isAllOnesConstant(RHS);
8747 EVT VT = N->getValueType(0);
8748 SDLoc DL(N);
8749
8750 // Match icmp_eq/ne(bitcast(icmp(X,Y)),0/-1) reduction patterns, and
8751 // change the outer compare to a i128 compare. This will normally
8752 // allow the reduction to be recognized in adjustICmp128, and even if
8753 // not, the i128 compare will still generate better code.
8754 if ((CC == ISD::SETNE || CC == ISD::SETEQ) && (CmpNull || CmpAllOnes)) {
8755 SDValue Src = peekThroughBitcasts(LHS);
8756 if (Src.getOpcode() == ISD::SETCC &&
8757 Src.getValueType().isFixedLengthVector() &&
8758 Src.getValueType().getScalarType() == MVT::i1) {
8759 EVT CmpVT = Src.getOperand(0).getValueType();
8760 if (CmpVT.getSizeInBits() == 128) {
8761 EVT IntVT = CmpVT.changeVectorElementTypeToInteger();
8762 SDValue LHS =
8763 DAG.getBitcast(MVT::i128, DAG.getSExtOrTrunc(Src, DL, IntVT));
8764 SDValue RHS = CmpNull ? DAG.getConstant(0, DL, MVT::i128)
8765 : DAG.getAllOnesConstant(DL, MVT::i128);
8766 return DAG.getNode(ISD::SETCC, DL, VT, LHS, RHS, N->getOperand(2),
8767 N->getFlags());
8768 }
8769 }
8770 }
8771
8772 return SDValue();
8773}
8774
8775static std::pair<SDValue, int> findCCUse(const SDValue &Val,
8776 unsigned Depth = 0) {
8777 // Limit depth of potentially exponential walk.
8778 if (Depth > 5)
8779 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8780
8781 switch (Val.getOpcode()) {
8782 default:
8783 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8784 case SystemZISD::IPM:
8785 if (Val.getOperand(0).getOpcode() == SystemZISD::CLC ||
8786 Val.getOperand(0).getOpcode() == SystemZISD::STRCMP)
8787 return std::make_pair(Val.getOperand(0), SystemZ::CCMASK_ICMP);
8788 return std::make_pair(Val.getOperand(0), SystemZ::CCMASK_ANY);
8789 case SystemZISD::SELECT_CCMASK: {
8790 SDValue Op4CCReg = Val.getOperand(4);
8791 if (Op4CCReg.getOpcode() == SystemZISD::ICMP ||
8792 Op4CCReg.getOpcode() == SystemZISD::TM) {
8793 auto [OpCC, OpCCValid] = findCCUse(Op4CCReg.getOperand(0), Depth + 1);
8794 if (OpCC != SDValue())
8795 return std::make_pair(OpCC, OpCCValid);
8796 }
8797 auto *CCValid = dyn_cast<ConstantSDNode>(Val.getOperand(2));
8798 if (!CCValid)
8799 return std::make_pair(SDValue(), SystemZ::CCMASK_NONE);
8800 int CCValidVal = CCValid->getZExtValue();
8801 return std::make_pair(Op4CCReg, CCValidVal);
8802 }
8803 case ISD::ADD:
8804 case ISD::AND:
8805 case ISD::OR:
8806 case ISD::XOR:
8807 case ISD::SHL:
8808 case ISD::SRA:
8809 case ISD::SRL:
8810 auto [Op0CC, Op0CCValid] = findCCUse(Val.getOperand(0), Depth + 1);
8811 if (Op0CC != SDValue())
8812 return std::make_pair(Op0CC, Op0CCValid);
8813 return findCCUse(Val.getOperand(1), Depth + 1);
8814 }
8815}
8816
8817static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask,
8818 SelectionDAG &DAG);
8819
8821 SelectionDAG &DAG) {
8822 SDLoc DL(Val);
8823 auto Opcode = Val.getOpcode();
8824 switch (Opcode) {
8825 default:
8826 return {};
8827 case ISD::Constant:
8828 return {Val, Val, Val, Val};
8829 case SystemZISD::IPM: {
8830 SDValue IPMOp0 = Val.getOperand(0);
8831 if (IPMOp0 != CC)
8832 return {};
8833 SmallVector<SDValue, 4> ShiftedCCVals;
8834 for (auto CC : {0, 1, 2, 3})
8835 ShiftedCCVals.emplace_back(
8836 DAG.getConstant((CC << SystemZ::IPM_CC), DL, MVT::i32));
8837 return ShiftedCCVals;
8838 }
8839 case SystemZISD::SELECT_CCMASK: {
8840 SDValue TrueVal = Val.getOperand(0), FalseVal = Val.getOperand(1);
8841 auto *CCValid = dyn_cast<ConstantSDNode>(Val.getOperand(2));
8842 auto *CCMask = dyn_cast<ConstantSDNode>(Val.getOperand(3));
8843 if (!CCValid || !CCMask)
8844 return {};
8845
8846 int CCValidVal = CCValid->getZExtValue();
8847 int CCMaskVal = CCMask->getZExtValue();
8848 // Pruning search tree early - Moving CC test and combineCCMask ahead of
8849 // recursive call to simplifyAssumingCCVal.
8850 SDValue Op4CCReg = Val.getOperand(4);
8851 if (Op4CCReg != CC)
8852 combineCCMask(Op4CCReg, CCValidVal, CCMaskVal, DAG);
8853 if (Op4CCReg != CC)
8854 return {};
8855 const auto &&TrueSDVals = simplifyAssumingCCVal(TrueVal, CC, DAG);
8856 const auto &&FalseSDVals = simplifyAssumingCCVal(FalseVal, CC, DAG);
8857 if (TrueSDVals.empty() || FalseSDVals.empty())
8858 return {};
8859 SmallVector<SDValue, 4> MergedSDVals;
8860 for (auto &CCVal : {0, 1, 2, 3})
8861 MergedSDVals.emplace_back(((CCMaskVal & (1 << (3 - CCVal))) != 0)
8862 ? TrueSDVals[CCVal]
8863 : FalseSDVals[CCVal]);
8864 return MergedSDVals;
8865 }
8866 case ISD::ADD:
8867 case ISD::AND:
8868 case ISD::OR:
8869 case ISD::XOR:
8870 case ISD::SRA:
8871 // Avoid introducing CC spills (because ADD/AND/OR/XOR/SRA
8872 // would clobber CC).
8873 if (!Val.hasOneUse())
8874 return {};
8875 [[fallthrough]];
8876 case ISD::SHL:
8877 case ISD::SRL:
8878 SDValue Op0 = Val.getOperand(0), Op1 = Val.getOperand(1);
8879 const auto &&Op0SDVals = simplifyAssumingCCVal(Op0, CC, DAG);
8880 const auto &&Op1SDVals = simplifyAssumingCCVal(Op1, CC, DAG);
8881 if (Op0SDVals.empty() || Op1SDVals.empty())
8882 return {};
8883 SmallVector<SDValue, 4> BinaryOpSDVals;
8884 for (auto CCVal : {0, 1, 2, 3})
8885 BinaryOpSDVals.emplace_back(DAG.getNode(
8886 Opcode, DL, Val.getValueType(), Op0SDVals[CCVal], Op1SDVals[CCVal]));
8887 return BinaryOpSDVals;
8888 }
8889}
8890
8891static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask,
8892 SelectionDAG &DAG) {
8893 // We have a SELECT_CCMASK or BR_CCMASK comparing the condition code
8894 // set by the CCReg instruction using the CCValid / CCMask masks,
8895 // If the CCReg instruction is itself a ICMP / TM testing the condition
8896 // code set by some other instruction, see whether we can directly
8897 // use that condition code.
8898 auto *CCNode = CCReg.getNode();
8899 if (!CCNode)
8900 return false;
8901
8902 if (CCNode->getOpcode() == SystemZISD::TM) {
8903 if (CCValid != SystemZ::CCMASK_TM)
8904 return false;
8905 auto emulateTMCCMask = [](const SDValue &Op0Val, const SDValue &Op1Val) {
8906 auto *Op0Node = dyn_cast<ConstantSDNode>(Op0Val.getNode());
8907 auto *Op1Node = dyn_cast<ConstantSDNode>(Op1Val.getNode());
8908 if (!Op0Node || !Op1Node)
8909 return -1;
8910 auto Op0APVal = Op0Node->getAPIntValue();
8911 auto Op1APVal = Op1Node->getAPIntValue();
8912 auto Result = Op0APVal & Op1APVal;
8913 bool AllOnes = Result == Op1APVal;
8914 bool AllZeros = Result == 0;
8915 bool IsLeftMostBitSet = Result[Op1APVal.getActiveBits() - 1] != 0;
8916 return AllZeros ? 0 : AllOnes ? 3 : IsLeftMostBitSet ? 2 : 1;
8917 };
8918 SDValue Op0 = CCNode->getOperand(0);
8919 SDValue Op1 = CCNode->getOperand(1);
8920 auto [Op0CC, Op0CCValid] = findCCUse(Op0);
8921 if (Op0CC == SDValue())
8922 return false;
8923 const auto &&Op0SDVals = simplifyAssumingCCVal(Op0, Op0CC, DAG);
8924 const auto &&Op1SDVals = simplifyAssumingCCVal(Op1, Op0CC, DAG);
8925 if (Op0SDVals.empty() || Op1SDVals.empty())
8926 return false;
8927 int NewCCMask = 0;
8928 for (auto CC : {0, 1, 2, 3}) {
8929 auto CCVal = emulateTMCCMask(Op0SDVals[CC], Op1SDVals[CC]);
8930 if (CCVal < 0)
8931 return false;
8932 NewCCMask <<= 1;
8933 NewCCMask |= (CCMask & (1 << (3 - CCVal))) != 0;
8934 }
8935 NewCCMask &= Op0CCValid;
8936 CCReg = Op0CC;
8937 CCMask = NewCCMask;
8938 CCValid = Op0CCValid;
8939 return true;
8940 }
8941 if (CCNode->getOpcode() != SystemZISD::ICMP ||
8942 CCValid != SystemZ::CCMASK_ICMP)
8943 return false;
8944
8945 SDValue CmpOp0 = CCNode->getOperand(0);
8946 SDValue CmpOp1 = CCNode->getOperand(1);
8947 SDValue CmpOp2 = CCNode->getOperand(2);
8948 auto [Op0CC, Op0CCValid] = findCCUse(CmpOp0);
8949 if (Op0CC != SDValue()) {
8950 const auto &&Op0SDVals = simplifyAssumingCCVal(CmpOp0, Op0CC, DAG);
8951 const auto &&Op1SDVals = simplifyAssumingCCVal(CmpOp1, Op0CC, DAG);
8952 if (Op0SDVals.empty() || Op1SDVals.empty())
8953 return false;
8954
8955 auto *CmpType = dyn_cast<ConstantSDNode>(CmpOp2);
8956 auto CmpTypeVal = CmpType->getZExtValue();
8957 const auto compareCCSigned = [&CmpTypeVal](const SDValue &Op0Val,
8958 const SDValue &Op1Val) {
8959 auto *Op0Node = dyn_cast<ConstantSDNode>(Op0Val.getNode());
8960 auto *Op1Node = dyn_cast<ConstantSDNode>(Op1Val.getNode());
8961 if (!Op0Node || !Op1Node)
8962 return -1;
8963 auto Op0APVal = Op0Node->getAPIntValue();
8964 auto Op1APVal = Op1Node->getAPIntValue();
8965 if (CmpTypeVal == SystemZICMP::SignedOnly)
8966 return Op0APVal == Op1APVal ? 0 : Op0APVal.slt(Op1APVal) ? 1 : 2;
8967 return Op0APVal == Op1APVal ? 0 : Op0APVal.ult(Op1APVal) ? 1 : 2;
8968 };
8969 int NewCCMask = 0;
8970 for (auto CC : {0, 1, 2, 3}) {
8971 auto CCVal = compareCCSigned(Op0SDVals[CC], Op1SDVals[CC]);
8972 if (CCVal < 0)
8973 return false;
8974 NewCCMask <<= 1;
8975 NewCCMask |= (CCMask & (1 << (3 - CCVal))) != 0;
8976 }
8977 NewCCMask &= Op0CCValid;
8978 CCMask = NewCCMask;
8979 CCReg = Op0CC;
8980 CCValid = Op0CCValid;
8981 return true;
8982 }
8983
8984 return false;
8985}
8986
8987// Merging versus split in multiple branches cost.
8990 const Value *Lhs,
8991 const Value *Rhs,
8992 const Function *) const {
8993 const auto isFlagOutOpCC = [](const Value *V) {
8994 using namespace llvm::PatternMatch;
8995 const Value *RHSVal;
8996 const APInt *RHSC;
8997 if (const auto *I = dyn_cast<Instruction>(V)) {
8998 // PatternMatch.h provides concise tree-based pattern match of llvm IR.
8999 if (match(I->getOperand(0), m_And(m_Value(RHSVal), m_APInt(RHSC))) ||
9000 match(I, m_Cmp(m_Value(RHSVal), m_APInt(RHSC)))) {
9001 if (const auto *CB = dyn_cast<CallBase>(RHSVal)) {
9002 if (CB->isInlineAsm()) {
9003 const InlineAsm *IA = cast<InlineAsm>(CB->getCalledOperand());
9004 return IA && IA->getConstraintString().contains("{@cc}");
9005 }
9006 }
9007 }
9008 }
9009 return false;
9010 };
9011 // Pattern (ICmp %asm) or (ICmp (And %asm)).
9012 // Cost of longest dependency chain (ICmp, And) is 2. CostThreshold or
9013 // BaseCost can be set >=2. If cost of instruction <= CostThreshold
9014 // conditionals will be merged or else conditionals will be split.
9015 if (isFlagOutOpCC(Lhs) && isFlagOutOpCC(Rhs))
9016 return {3, 0, -1};
9017 // Default.
9018 return {-1, -1, -1};
9019}
9020
9021SDValue SystemZTargetLowering::combineBR_CCMASK(SDNode *N,
9022 DAGCombinerInfo &DCI) const {
9023 SelectionDAG &DAG = DCI.DAG;
9024
9025 // Combine BR_CCMASK (ICMP (SELECT_CCMASK)) into a single BR_CCMASK.
9026 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(1));
9027 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(2));
9028 if (!CCValid || !CCMask)
9029 return SDValue();
9030
9031 int CCValidVal = CCValid->getZExtValue();
9032 int CCMaskVal = CCMask->getZExtValue();
9033 SDValue Chain = N->getOperand(0);
9034 SDValue CCReg = N->getOperand(4);
9035 // If combineCMask was able to merge or simplify ccvalid or ccmask, re-emit
9036 // the modified BR_CCMASK with the new values.
9037 // In order to avoid conditional branches with full or empty cc masks, do not
9038 // do this if ccmask is 0 or equal to ccvalid.
9039 if (combineCCMask(CCReg, CCValidVal, CCMaskVal, DAG) && CCMaskVal != 0 &&
9040 CCMaskVal != CCValidVal)
9041 return DAG.getNode(SystemZISD::BR_CCMASK, SDLoc(N), N->getValueType(0),
9042 Chain,
9043 DAG.getTargetConstant(CCValidVal, SDLoc(N), MVT::i32),
9044 DAG.getTargetConstant(CCMaskVal, SDLoc(N), MVT::i32),
9045 N->getOperand(3), CCReg);
9046 return SDValue();
9047}
9048
9049SDValue SystemZTargetLowering::combineSELECT_CCMASK(
9050 SDNode *N, DAGCombinerInfo &DCI) const {
9051 SelectionDAG &DAG = DCI.DAG;
9052
9053 // Combine SELECT_CCMASK (ICMP (SELECT_CCMASK)) into a single SELECT_CCMASK.
9054 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(2));
9055 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(3));
9056 if (!CCValid || !CCMask)
9057 return SDValue();
9058
9059 int CCValidVal = CCValid->getZExtValue();
9060 int CCMaskVal = CCMask->getZExtValue();
9061 SDValue CCReg = N->getOperand(4);
9062
9063 bool IsCombinedCCReg = combineCCMask(CCReg, CCValidVal, CCMaskVal, DAG);
9064
9065 // Populate SDVals vector for each condition code ccval for given Val, which
9066 // can again be another nested select_ccmask with the same CC.
9067 const auto constructCCSDValsFromSELECT = [&CCReg](SDValue &Val) {
9068 if (Val.getOpcode() == SystemZISD::SELECT_CCMASK) {
9070 if (Val.getOperand(4) != CCReg)
9071 return SmallVector<SDValue, 4>{};
9072 SDValue TrueVal = Val.getOperand(0), FalseVal = Val.getOperand(1);
9073 auto *CCMask = dyn_cast<ConstantSDNode>(Val.getOperand(3));
9074 if (!CCMask)
9075 return SmallVector<SDValue, 4>{};
9076
9077 int CCMaskVal = CCMask->getZExtValue();
9078 for (auto &CC : {0, 1, 2, 3})
9079 Res.emplace_back(((CCMaskVal & (1 << (3 - CC))) != 0) ? TrueVal
9080 : FalseVal);
9081 return Res;
9082 }
9083 return SmallVector<SDValue, 4>{Val, Val, Val, Val};
9084 };
9085 // Attempting to optimize TrueVal/FalseVal in outermost select_ccmask either
9086 // with CCReg found by combineCCMask or original CCReg.
9087 SDValue TrueVal = N->getOperand(0);
9088 SDValue FalseVal = N->getOperand(1);
9089 auto &&TrueSDVals = simplifyAssumingCCVal(TrueVal, CCReg, DAG);
9090 auto &&FalseSDVals = simplifyAssumingCCVal(FalseVal, CCReg, DAG);
9091 // TrueSDVals/FalseSDVals might be empty in case of non-constant
9092 // TrueVal/FalseVal for select_ccmask, which can not be optimized further.
9093 if (TrueSDVals.empty())
9094 TrueSDVals = constructCCSDValsFromSELECT(TrueVal);
9095 if (FalseSDVals.empty())
9096 FalseSDVals = constructCCSDValsFromSELECT(FalseVal);
9097 if (!TrueSDVals.empty() && !FalseSDVals.empty()) {
9098 SmallSet<SDValue, 4> MergedSDValsSet;
9099 // Ignoring CC values outside CCValiid.
9100 for (auto CC : {0, 1, 2, 3}) {
9101 if ((CCValidVal & ((1 << (3 - CC)))) != 0)
9102 MergedSDValsSet.insert(((CCMaskVal & (1 << (3 - CC))) != 0)
9103 ? TrueSDVals[CC]
9104 : FalseSDVals[CC]);
9105 }
9106 if (MergedSDValsSet.size() == 1)
9107 return *MergedSDValsSet.begin();
9108 if (MergedSDValsSet.size() == 2) {
9109 auto BeginIt = MergedSDValsSet.begin();
9110 SDValue NewTrueVal = *BeginIt, NewFalseVal = *next(BeginIt);
9111 if (NewTrueVal == FalseVal || NewFalseVal == TrueVal)
9112 std::swap(NewTrueVal, NewFalseVal);
9113 int NewCCMask = 0;
9114 for (auto CC : {0, 1, 2, 3}) {
9115 NewCCMask <<= 1;
9116 NewCCMask |= ((CCMaskVal & (1 << (3 - CC))) != 0)
9117 ? (TrueSDVals[CC] == NewTrueVal)
9118 : (FalseSDVals[CC] == NewTrueVal);
9119 }
9120 CCMaskVal = NewCCMask;
9121 CCMaskVal &= CCValidVal;
9122 TrueVal = NewTrueVal;
9123 FalseVal = NewFalseVal;
9124 IsCombinedCCReg = true;
9125 }
9126 }
9127 // If the condition is trivially false or trivially true after
9128 // combineCCMask, just collapse this SELECT_CCMASK to the indicated value
9129 // (possibly modified by constructCCSDValsFromSELECT).
9130 if (CCMaskVal == 0)
9131 return FalseVal;
9132 if (CCMaskVal == CCValidVal)
9133 return TrueVal;
9134
9135 if (IsCombinedCCReg)
9136 return DAG.getNode(
9137 SystemZISD::SELECT_CCMASK, SDLoc(N), N->getValueType(0), TrueVal,
9138 FalseVal, DAG.getTargetConstant(CCValidVal, SDLoc(N), MVT::i32),
9139 DAG.getTargetConstant(CCMaskVal, SDLoc(N), MVT::i32), CCReg);
9140
9141 return SDValue();
9142}
9143
9144SDValue SystemZTargetLowering::combineGET_CCMASK(
9145 SDNode *N, DAGCombinerInfo &DCI) const {
9146
9147 // Optimize away GET_CCMASK (SELECT_CCMASK) if the CC masks are compatible
9148 auto *CCValid = dyn_cast<ConstantSDNode>(N->getOperand(1));
9149 auto *CCMask = dyn_cast<ConstantSDNode>(N->getOperand(2));
9150 if (!CCValid || !CCMask)
9151 return SDValue();
9152 int CCValidVal = CCValid->getZExtValue();
9153 int CCMaskVal = CCMask->getZExtValue();
9154
9155 SDValue Select = N->getOperand(0);
9156 if (Select->getOpcode() == ISD::TRUNCATE)
9157 Select = Select->getOperand(0);
9158 if (Select->getOpcode() != SystemZISD::SELECT_CCMASK)
9159 return SDValue();
9160
9161 auto *SelectCCValid = dyn_cast<ConstantSDNode>(Select->getOperand(2));
9162 auto *SelectCCMask = dyn_cast<ConstantSDNode>(Select->getOperand(3));
9163 if (!SelectCCValid || !SelectCCMask)
9164 return SDValue();
9165 int SelectCCValidVal = SelectCCValid->getZExtValue();
9166 int SelectCCMaskVal = SelectCCMask->getZExtValue();
9167
9168 auto *TrueVal = dyn_cast<ConstantSDNode>(Select->getOperand(0));
9169 auto *FalseVal = dyn_cast<ConstantSDNode>(Select->getOperand(1));
9170 if (!TrueVal || !FalseVal)
9171 return SDValue();
9172 if (TrueVal->getZExtValue() == 1 && FalseVal->getZExtValue() == 0)
9173 ;
9174 else if (TrueVal->getZExtValue() == 0 && FalseVal->getZExtValue() == 1)
9175 SelectCCMaskVal ^= SelectCCValidVal;
9176 else
9177 return SDValue();
9178
9179 if (SelectCCValidVal & ~CCValidVal)
9180 return SDValue();
9181 if (SelectCCMaskVal != (CCMaskVal & SelectCCValidVal))
9182 return SDValue();
9183
9184 return Select->getOperand(4);
9185}
9186
9187SDValue SystemZTargetLowering::combineIntDIVREM(
9188 SDNode *N, DAGCombinerInfo &DCI) const {
9189 SelectionDAG &DAG = DCI.DAG;
9190 EVT VT = N->getValueType(0);
9191 // In the case where the divisor is a vector of constants a cheaper
9192 // sequence of instructions can replace the divide. BuildSDIV is called to
9193 // do this during DAG combining, but it only succeeds when it can build a
9194 // multiplication node. The only option for SystemZ is ISD::SMUL_LOHI, and
9195 // since it is not Legal but Custom it can only happen before
9196 // legalization. Therefore we must scalarize this early before Combine
9197 // 1. For widened vectors, this is already the result of type legalization.
9198 if (DCI.Level == BeforeLegalizeTypes && VT.isVector() && isTypeLegal(VT) &&
9199 DAG.isConstantIntBuildVectorOrConstantInt(N->getOperand(1)))
9200 return DAG.UnrollVectorOp(N);
9201 return SDValue();
9202}
9203
9204
9205// Transform a right shift of a multiply-and-add into a multiply-and-add-high.
9206// This is closely modeled after the common-code combineShiftToMULH.
9207SDValue SystemZTargetLowering::combineShiftToMulAddHigh(
9208 SDNode *N, DAGCombinerInfo &DCI) const {
9209 SelectionDAG &DAG = DCI.DAG;
9210 SDLoc DL(N);
9211
9212 assert((N->getOpcode() == ISD::SRL || N->getOpcode() == ISD::SRA) &&
9213 "SRL or SRA node is required here!");
9214
9215 if (!Subtarget.hasVector())
9216 return SDValue();
9217
9218 // Check the shift amount. Proceed with the transformation if the shift
9219 // amount is constant.
9220 ConstantSDNode *ShiftAmtSrc = isConstOrConstSplat(N->getOperand(1));
9221 if (!ShiftAmtSrc)
9222 return SDValue();
9223
9224 // The operation feeding into the shift must be an add.
9225 SDValue ShiftOperand = N->getOperand(0);
9226 if (ShiftOperand.getOpcode() != ISD::ADD)
9227 return SDValue();
9228
9229 // One operand of the add must be a multiply.
9230 SDValue MulOp = ShiftOperand.getOperand(0);
9231 SDValue AddOp = ShiftOperand.getOperand(1);
9232 if (MulOp.getOpcode() != ISD::MUL) {
9233 if (AddOp.getOpcode() != ISD::MUL)
9234 return SDValue();
9235 std::swap(MulOp, AddOp);
9236 }
9237
9238 // All operands must be equivalent extend nodes.
9239 SDValue LeftOp = MulOp.getOperand(0);
9240 SDValue RightOp = MulOp.getOperand(1);
9241
9242 bool IsSignExt = LeftOp.getOpcode() == ISD::SIGN_EXTEND;
9243 bool IsZeroExt = LeftOp.getOpcode() == ISD::ZERO_EXTEND;
9244
9245 if (!IsSignExt && !IsZeroExt)
9246 return SDValue();
9247
9248 EVT NarrowVT = LeftOp.getOperand(0).getValueType();
9249 unsigned NarrowVTSize = NarrowVT.getScalarSizeInBits();
9250
9251 SDValue MulhRightOp;
9252 if (ConstantSDNode *Constant = isConstOrConstSplat(RightOp)) {
9253 unsigned ActiveBits = IsSignExt
9254 ? Constant->getAPIntValue().getSignificantBits()
9255 : Constant->getAPIntValue().getActiveBits();
9256 if (ActiveBits > NarrowVTSize)
9257 return SDValue();
9258 MulhRightOp = DAG.getConstant(
9259 Constant->getAPIntValue().trunc(NarrowVT.getScalarSizeInBits()), DL,
9260 NarrowVT);
9261 } else {
9262 if (LeftOp.getOpcode() != RightOp.getOpcode())
9263 return SDValue();
9264 // Check that the two extend nodes are the same type.
9265 if (NarrowVT != RightOp.getOperand(0).getValueType())
9266 return SDValue();
9267 MulhRightOp = RightOp.getOperand(0);
9268 }
9269
9270 SDValue MulhAddOp;
9271 if (ConstantSDNode *Constant = isConstOrConstSplat(AddOp)) {
9272 unsigned ActiveBits = IsSignExt
9273 ? Constant->getAPIntValue().getSignificantBits()
9274 : Constant->getAPIntValue().getActiveBits();
9275 if (ActiveBits > NarrowVTSize)
9276 return SDValue();
9277 MulhAddOp = DAG.getConstant(
9278 Constant->getAPIntValue().trunc(NarrowVT.getScalarSizeInBits()), DL,
9279 NarrowVT);
9280 } else {
9281 if (LeftOp.getOpcode() != AddOp.getOpcode())
9282 return SDValue();
9283 // Check that the two extend nodes are the same type.
9284 if (NarrowVT != AddOp.getOperand(0).getValueType())
9285 return SDValue();
9286 MulhAddOp = AddOp.getOperand(0);
9287 }
9288
9289 EVT WideVT = LeftOp.getValueType();
9290 // Proceed with the transformation if the wide types match.
9291 assert((WideVT == RightOp.getValueType()) &&
9292 "Cannot have a multiply node with two different operand types.");
9293 assert((WideVT == AddOp.getValueType()) &&
9294 "Cannot have an add node with two different operand types.");
9295
9296 // Proceed with the transformation if the wide type is twice as large
9297 // as the narrow type.
9298 if (WideVT.getScalarSizeInBits() != 2 * NarrowVTSize)
9299 return SDValue();
9300
9301 // Check the shift amount with the narrow type size.
9302 // Proceed with the transformation if the shift amount is the width
9303 // of the narrow type.
9304 unsigned ShiftAmt = ShiftAmtSrc->getZExtValue();
9305 if (ShiftAmt != NarrowVTSize)
9306 return SDValue();
9307
9308 // Proceed if we support the multiply-and-add-high operation.
9309 if (!(NarrowVT == MVT::v16i8 || NarrowVT == MVT::v8i16 ||
9310 NarrowVT == MVT::v4i32 ||
9311 (Subtarget.hasVectorEnhancements3() &&
9312 (NarrowVT == MVT::v2i64 || NarrowVT == MVT::i128))))
9313 return SDValue();
9314
9315 // Emit the VMAH (signed) or VMALH (unsigned) operation.
9316 SDValue Result = DAG.getNode(IsSignExt ? SystemZISD::VMAH : SystemZISD::VMALH,
9317 DL, NarrowVT, LeftOp.getOperand(0),
9318 MulhRightOp, MulhAddOp);
9319 bool IsSigned = N->getOpcode() == ISD::SRA;
9320 return DAG.getExtOrTrunc(IsSigned, Result, DL, WideVT);
9321}
9322
9323// Op is an operand of a multiplication. Check whether this can be folded
9324// into an even/odd widening operation; if so, return the opcode to be used
9325// and update Op to the appropriate sub-operand. Note that the caller must
9326// verify that *both* operands of the multiplication support the operation.
9328 const SystemZSubtarget &Subtarget,
9329 SDValue &Op) {
9330 EVT VT = Op.getValueType();
9331
9332 // Check for (sign/zero_extend_vector_inreg (vector_shuffle)) corresponding
9333 // to selecting the even or odd vector elements.
9334 if (VT.isVector() && DAG.getTargetLoweringInfo().isTypeLegal(VT) &&
9335 (Op.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG ||
9336 Op.getOpcode() == ISD::ZERO_EXTEND_VECTOR_INREG)) {
9337 bool IsSigned = Op.getOpcode() == ISD::SIGN_EXTEND_VECTOR_INREG;
9338 unsigned NumElts = VT.getVectorNumElements();
9339 Op = Op.getOperand(0);
9340 if (Op.getValueType().getVectorNumElements() == 2 * NumElts &&
9341 Op.getOpcode() == ISD::VECTOR_SHUFFLE) {
9343 ArrayRef<int> ShuffleMask = SVN->getMask();
9344 bool CanUseEven = true, CanUseOdd = true;
9345 for (unsigned Elt = 0; Elt < NumElts; Elt++) {
9346 if (ShuffleMask[Elt] == -1)
9347 continue;
9348 if (unsigned(ShuffleMask[Elt]) != 2 * Elt)
9349 CanUseEven = false;
9350 if (unsigned(ShuffleMask[Elt]) != 2 * Elt + 1)
9351 CanUseOdd = false;
9352 }
9353 Op = Op.getOperand(0);
9354 if (CanUseEven)
9355 return IsSigned ? SystemZISD::VME : SystemZISD::VMLE;
9356 if (CanUseOdd)
9357 return IsSigned ? SystemZISD::VMO : SystemZISD::VMLO;
9358 }
9359 }
9360
9361 // For z17, we can also support the v2i64->i128 case, which looks like
9362 // (sign/zero_extend (extract_vector_elt X 0/1))
9363 if (VT == MVT::i128 && Subtarget.hasVectorEnhancements3() &&
9364 (Op.getOpcode() == ISD::SIGN_EXTEND ||
9365 Op.getOpcode() == ISD::ZERO_EXTEND)) {
9366 bool IsSigned = Op.getOpcode() == ISD::SIGN_EXTEND;
9367 Op = Op.getOperand(0);
9368 if (Op.getOpcode() == ISD::EXTRACT_VECTOR_ELT &&
9369 Op.getOperand(0).getValueType() == MVT::v2i64 &&
9370 Op.getOperand(1).getOpcode() == ISD::Constant) {
9371 unsigned Elem = Op.getConstantOperandVal(1);
9372 Op = Op.getOperand(0);
9373 if (Elem == 0)
9374 return IsSigned ? SystemZISD::VME : SystemZISD::VMLE;
9375 if (Elem == 1)
9376 return IsSigned ? SystemZISD::VMO : SystemZISD::VMLO;
9377 }
9378 }
9379
9380 return 0;
9381}
9382
9383SDValue SystemZTargetLowering::combineMUL(
9384 SDNode *N, DAGCombinerInfo &DCI) const {
9385 SelectionDAG &DAG = DCI.DAG;
9386
9387 // Detect even/odd widening multiplication.
9388 SDValue Op0 = N->getOperand(0);
9389 SDValue Op1 = N->getOperand(1);
9390 unsigned OpcodeCand0 = detectEvenOddMultiplyOperand(DAG, Subtarget, Op0);
9391 unsigned OpcodeCand1 = detectEvenOddMultiplyOperand(DAG, Subtarget, Op1);
9392 if (OpcodeCand0 && OpcodeCand0 == OpcodeCand1)
9393 return DAG.getNode(OpcodeCand0, SDLoc(N), N->getValueType(0), Op0, Op1);
9394
9395 return SDValue();
9396}
9397
9398SDValue SystemZTargetLowering::combineINTRINSIC(
9399 SDNode *N, DAGCombinerInfo &DCI) const {
9400 SelectionDAG &DAG = DCI.DAG;
9401
9402 unsigned Id = N->getConstantOperandVal(1);
9403 switch (Id) {
9404 // VECTOR LOAD (RIGHTMOST) WITH LENGTH with a length operand of 15
9405 // or larger is simply a vector load.
9406 case Intrinsic::s390_vll:
9407 case Intrinsic::s390_vlrl:
9408 if (auto *C = dyn_cast<ConstantSDNode>(N->getOperand(2)))
9409 if (C->getZExtValue() >= 15)
9410 return DAG.getLoad(N->getValueType(0), SDLoc(N), N->getOperand(0),
9411 N->getOperand(3), MachinePointerInfo());
9412 break;
9413 // Likewise for VECTOR STORE (RIGHTMOST) WITH LENGTH.
9414 case Intrinsic::s390_vstl:
9415 case Intrinsic::s390_vstrl:
9416 if (auto *C = dyn_cast<ConstantSDNode>(N->getOperand(3)))
9417 if (C->getZExtValue() >= 15)
9418 return DAG.getStore(N->getOperand(0), SDLoc(N), N->getOperand(2),
9419 N->getOperand(4), MachinePointerInfo());
9420 break;
9421 }
9422
9423 return SDValue();
9424}
9425
9426SDValue SystemZTargetLowering::unwrapAddress(SDValue N) const {
9427 if (N->getOpcode() == SystemZISD::PCREL_WRAPPER)
9428 return N->getOperand(0);
9429 return N;
9430}
9431
9433 DAGCombinerInfo &DCI) const {
9434 switch(N->getOpcode()) {
9435 default: break;
9436 case ISD::ZERO_EXTEND: return combineZERO_EXTEND(N, DCI);
9437 case ISD::SIGN_EXTEND: return combineSIGN_EXTEND(N, DCI);
9438 case ISD::SIGN_EXTEND_INREG: return combineSIGN_EXTEND_INREG(N, DCI);
9439 case SystemZISD::MERGE_HIGH:
9440 case SystemZISD::MERGE_LOW: return combineMERGE(N, DCI);
9441 case ISD::LOAD: return combineLOAD(N, DCI);
9442 case ISD::STORE: return combineSTORE(N, DCI);
9443 case ISD::VECTOR_SHUFFLE: return combineVECTOR_SHUFFLE(N, DCI);
9444 case ISD::EXTRACT_VECTOR_ELT: return combineEXTRACT_VECTOR_ELT(N, DCI);
9445 case SystemZISD::JOIN_DWORDS: return combineJOIN_DWORDS(N, DCI);
9447 case ISD::FP_ROUND: return combineFP_ROUND(N, DCI);
9449 case ISD::FP_EXTEND: return combineFP_EXTEND(N, DCI);
9450 case ISD::SINT_TO_FP:
9451 case ISD::UINT_TO_FP: return combineINT_TO_FP(N, DCI);
9452 case ISD::FCOPYSIGN: return combineFCOPYSIGN(N, DCI);
9453 case ISD::BSWAP: return combineBSWAP(N, DCI);
9454 case ISD::SETCC: return combineSETCC(N, DCI);
9455 case SystemZISD::BR_CCMASK: return combineBR_CCMASK(N, DCI);
9456 case SystemZISD::SELECT_CCMASK: return combineSELECT_CCMASK(N, DCI);
9457 case SystemZISD::GET_CCMASK: return combineGET_CCMASK(N, DCI);
9458 case ISD::SRL:
9459 case ISD::SRA: return combineShiftToMulAddHigh(N, DCI);
9460 case ISD::MUL: return combineMUL(N, DCI);
9461 case ISD::SDIV:
9462 case ISD::UDIV:
9463 case ISD::SREM:
9464 case ISD::UREM: return combineIntDIVREM(N, DCI);
9466 case ISD::INTRINSIC_VOID: return combineINTRINSIC(N, DCI);
9467 }
9468
9469 return SDValue();
9470}
9471
9472// Return the demanded elements for the OpNo source operand of Op. DemandedElts
9473// are for Op.
9474static APInt getDemandedSrcElements(SDValue Op, const APInt &DemandedElts,
9475 unsigned OpNo) {
9476 EVT VT = Op.getValueType();
9477 unsigned NumElts = (VT.isVector() ? VT.getVectorNumElements() : 1);
9478 APInt SrcDemE;
9479 unsigned Opcode = Op.getOpcode();
9480 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9481 unsigned Id = Op.getConstantOperandVal(0);
9482 switch (Id) {
9483 case Intrinsic::s390_vpksh: // PACKS
9484 case Intrinsic::s390_vpksf:
9485 case Intrinsic::s390_vpksg:
9486 case Intrinsic::s390_vpkshs: // PACKS_CC
9487 case Intrinsic::s390_vpksfs:
9488 case Intrinsic::s390_vpksgs:
9489 case Intrinsic::s390_vpklsh: // PACKLS
9490 case Intrinsic::s390_vpklsf:
9491 case Intrinsic::s390_vpklsg:
9492 case Intrinsic::s390_vpklshs: // PACKLS_CC
9493 case Intrinsic::s390_vpklsfs:
9494 case Intrinsic::s390_vpklsgs:
9495 // VECTOR PACK truncates the elements of two source vectors into one.
9496 SrcDemE = DemandedElts;
9497 if (OpNo == 2)
9498 SrcDemE.lshrInPlace(NumElts / 2);
9499 SrcDemE = SrcDemE.trunc(NumElts / 2);
9500 break;
9501 // VECTOR UNPACK extends half the elements of the source vector.
9502 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9503 case Intrinsic::s390_vuphh:
9504 case Intrinsic::s390_vuphf:
9505 case Intrinsic::s390_vuplhb: // VECTOR UNPACK LOGICAL HIGH
9506 case Intrinsic::s390_vuplhh:
9507 case Intrinsic::s390_vuplhf:
9508 SrcDemE = APInt(NumElts * 2, 0);
9509 SrcDemE.insertBits(DemandedElts, 0);
9510 break;
9511 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9512 case Intrinsic::s390_vuplhw:
9513 case Intrinsic::s390_vuplf:
9514 case Intrinsic::s390_vupllb: // VECTOR UNPACK LOGICAL LOW
9515 case Intrinsic::s390_vupllh:
9516 case Intrinsic::s390_vupllf:
9517 SrcDemE = APInt(NumElts * 2, 0);
9518 SrcDemE.insertBits(DemandedElts, NumElts);
9519 break;
9520 case Intrinsic::s390_vpdi: {
9521 // VECTOR PERMUTE DWORD IMMEDIATE selects one element from each source.
9522 SrcDemE = APInt(NumElts, 0);
9523 if (!DemandedElts[OpNo - 1])
9524 break;
9525 unsigned Mask = Op.getConstantOperandVal(3);
9526 unsigned MaskBit = ((OpNo - 1) ? 1 : 4);
9527 // Demand input element 0 or 1, given by the mask bit value.
9528 SrcDemE.setBit((Mask & MaskBit)? 1 : 0);
9529 break;
9530 }
9531 case Intrinsic::s390_vsldb: {
9532 // VECTOR SHIFT LEFT DOUBLE BY BYTE
9533 assert(VT == MVT::v16i8 && "Unexpected type.");
9534 unsigned FirstIdx = Op.getConstantOperandVal(3);
9535 assert (FirstIdx > 0 && FirstIdx < 16 && "Unused operand.");
9536 unsigned NumSrc0Els = 16 - FirstIdx;
9537 SrcDemE = APInt(NumElts, 0);
9538 if (OpNo == 1) {
9539 APInt DemEls = DemandedElts.trunc(NumSrc0Els);
9540 SrcDemE.insertBits(DemEls, FirstIdx);
9541 } else {
9542 APInt DemEls = DemandedElts.lshr(NumSrc0Els);
9543 SrcDemE.insertBits(DemEls, 0);
9544 }
9545 break;
9546 }
9547 case Intrinsic::s390_vperm:
9548 SrcDemE = APInt::getAllOnes(NumElts);
9549 break;
9550 default:
9551 llvm_unreachable("Unhandled intrinsic.");
9552 break;
9553 }
9554 } else {
9555 switch (Opcode) {
9556 case SystemZISD::JOIN_DWORDS:
9557 // Scalar operand.
9558 SrcDemE = APInt(1, 1);
9559 break;
9560 case SystemZISD::SELECT_CCMASK:
9561 SrcDemE = DemandedElts;
9562 break;
9563 default:
9564 llvm_unreachable("Unhandled opcode.");
9565 break;
9566 }
9567 }
9568 return SrcDemE;
9569}
9570
9572 const APInt &DemandedElts,
9573 const SelectionDAG &DAG, unsigned Depth,
9574 unsigned OpNo) {
9575 APInt Src0DemE = getDemandedSrcElements(Op, DemandedElts, OpNo);
9576 APInt Src1DemE = getDemandedSrcElements(Op, DemandedElts, OpNo + 1);
9577 KnownBits LHSKnown =
9578 DAG.computeKnownBits(Op.getOperand(OpNo), Src0DemE, Depth + 1);
9579 KnownBits RHSKnown =
9580 DAG.computeKnownBits(Op.getOperand(OpNo + 1), Src1DemE, Depth + 1);
9581 Known = LHSKnown.intersectWith(RHSKnown);
9582}
9583
9584void
9587 const APInt &DemandedElts,
9588 const SelectionDAG &DAG,
9589 unsigned Depth) const {
9590 Known.resetAll();
9591
9592 // Intrinsic CC result is returned in the two low bits.
9593 unsigned Tmp0, Tmp1; // not used
9594 if (Op.getResNo() == 1 && isIntrinsicWithCC(Op, Tmp0, Tmp1)) {
9595 Known.Zero.setBitsFrom(2);
9596 return;
9597 }
9598 EVT VT = Op.getValueType();
9599 if (Op.getResNo() != 0 || VT == MVT::Untyped)
9600 return;
9601 assert (Known.getBitWidth() == VT.getScalarSizeInBits() &&
9602 "KnownBits does not match VT in bitwidth");
9603 assert ((!VT.isVector() ||
9604 (DemandedElts.getBitWidth() == VT.getVectorNumElements())) &&
9605 "DemandedElts does not match VT number of elements");
9606 unsigned BitWidth = Known.getBitWidth();
9607 unsigned Opcode = Op.getOpcode();
9608 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9609 bool IsLogical = false;
9610 unsigned Id = Op.getConstantOperandVal(0);
9611 switch (Id) {
9612 case Intrinsic::s390_vpksh: // PACKS
9613 case Intrinsic::s390_vpksf:
9614 case Intrinsic::s390_vpksg:
9615 case Intrinsic::s390_vpkshs: // PACKS_CC
9616 case Intrinsic::s390_vpksfs:
9617 case Intrinsic::s390_vpksgs:
9618 case Intrinsic::s390_vpklsh: // PACKLS
9619 case Intrinsic::s390_vpklsf:
9620 case Intrinsic::s390_vpklsg:
9621 case Intrinsic::s390_vpklshs: // PACKLS_CC
9622 case Intrinsic::s390_vpklsfs:
9623 case Intrinsic::s390_vpklsgs:
9624 case Intrinsic::s390_vpdi:
9625 case Intrinsic::s390_vsldb:
9626 case Intrinsic::s390_vperm:
9627 computeKnownBitsBinOp(Op, Known, DemandedElts, DAG, Depth, 1);
9628 break;
9629 case Intrinsic::s390_vuplhb: // VECTOR UNPACK LOGICAL HIGH
9630 case Intrinsic::s390_vuplhh:
9631 case Intrinsic::s390_vuplhf:
9632 case Intrinsic::s390_vupllb: // VECTOR UNPACK LOGICAL LOW
9633 case Intrinsic::s390_vupllh:
9634 case Intrinsic::s390_vupllf:
9635 IsLogical = true;
9636 [[fallthrough]];
9637 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9638 case Intrinsic::s390_vuphh:
9639 case Intrinsic::s390_vuphf:
9640 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9641 case Intrinsic::s390_vuplhw:
9642 case Intrinsic::s390_vuplf: {
9643 SDValue SrcOp = Op.getOperand(1);
9644 APInt SrcDemE = getDemandedSrcElements(Op, DemandedElts, 0);
9645 Known = DAG.computeKnownBits(SrcOp, SrcDemE, Depth + 1);
9646 if (IsLogical) {
9647 Known = Known.zext(BitWidth);
9648 } else
9649 Known = Known.sext(BitWidth);
9650 break;
9651 }
9652 default:
9653 break;
9654 }
9655 } else {
9656 switch (Opcode) {
9657 case SystemZISD::JOIN_DWORDS:
9658 case SystemZISD::SELECT_CCMASK:
9659 computeKnownBitsBinOp(Op, Known, DemandedElts, DAG, Depth, 0);
9660 break;
9661 case SystemZISD::REPLICATE: {
9662 SDValue SrcOp = Op.getOperand(0);
9663 Known = DAG.computeKnownBits(SrcOp, Depth + 1);
9664 if (Known.getBitWidth() < BitWidth && isa<ConstantSDNode>(SrcOp))
9665 Known = Known.sext(BitWidth); // VREPI sign extends the immedate.
9666 break;
9667 }
9668 default:
9669 break;
9670 }
9671 }
9672
9673 // Known has the width of the source operand(s). Adjust if needed to match
9674 // the passed bitwidth.
9675 if (Known.getBitWidth() != BitWidth)
9676 Known = Known.anyextOrTrunc(BitWidth);
9677}
9678
9679static unsigned computeNumSignBitsBinOp(SDValue Op, const APInt &DemandedElts,
9680 const SelectionDAG &DAG, unsigned Depth,
9681 unsigned OpNo) {
9682 APInt Src0DemE = getDemandedSrcElements(Op, DemandedElts, OpNo);
9683 unsigned LHS = DAG.ComputeNumSignBits(Op.getOperand(OpNo), Src0DemE, Depth + 1);
9684 if (LHS == 1) return 1; // Early out.
9685 APInt Src1DemE = getDemandedSrcElements(Op, DemandedElts, OpNo + 1);
9686 unsigned RHS = DAG.ComputeNumSignBits(Op.getOperand(OpNo + 1), Src1DemE, Depth + 1);
9687 if (RHS == 1) return 1; // Early out.
9688 unsigned Common = std::min(LHS, RHS);
9689 unsigned SrcBitWidth = Op.getOperand(OpNo).getScalarValueSizeInBits();
9690 EVT VT = Op.getValueType();
9691 unsigned VTBits = VT.getScalarSizeInBits();
9692 if (SrcBitWidth > VTBits) { // PACK
9693 unsigned SrcExtraBits = SrcBitWidth - VTBits;
9694 if (Common > SrcExtraBits)
9695 return (Common - SrcExtraBits);
9696 return 1;
9697 }
9698 assert (SrcBitWidth == VTBits && "Expected operands of same bitwidth.");
9699 return Common;
9700}
9701
9702unsigned
9704 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
9705 unsigned Depth) const {
9706 if (Op.getResNo() != 0)
9707 return 1;
9708 unsigned Opcode = Op.getOpcode();
9709 if (Opcode == ISD::INTRINSIC_WO_CHAIN) {
9710 unsigned Id = Op.getConstantOperandVal(0);
9711 switch (Id) {
9712 case Intrinsic::s390_vpksh: // PACKS
9713 case Intrinsic::s390_vpksf:
9714 case Intrinsic::s390_vpksg:
9715 case Intrinsic::s390_vpkshs: // PACKS_CC
9716 case Intrinsic::s390_vpksfs:
9717 case Intrinsic::s390_vpksgs:
9718 case Intrinsic::s390_vpklsh: // PACKLS
9719 case Intrinsic::s390_vpklsf:
9720 case Intrinsic::s390_vpklsg:
9721 case Intrinsic::s390_vpklshs: // PACKLS_CC
9722 case Intrinsic::s390_vpklsfs:
9723 case Intrinsic::s390_vpklsgs:
9724 case Intrinsic::s390_vpdi:
9725 case Intrinsic::s390_vsldb:
9726 case Intrinsic::s390_vperm:
9727 return computeNumSignBitsBinOp(Op, DemandedElts, DAG, Depth, 1);
9728 case Intrinsic::s390_vuphb: // VECTOR UNPACK HIGH
9729 case Intrinsic::s390_vuphh:
9730 case Intrinsic::s390_vuphf:
9731 case Intrinsic::s390_vuplb: // VECTOR UNPACK LOW
9732 case Intrinsic::s390_vuplhw:
9733 case Intrinsic::s390_vuplf: {
9734 SDValue PackedOp = Op.getOperand(1);
9735 APInt SrcDemE = getDemandedSrcElements(Op, DemandedElts, 1);
9736 unsigned Tmp = DAG.ComputeNumSignBits(PackedOp, SrcDemE, Depth + 1);
9737 EVT VT = Op.getValueType();
9738 unsigned VTBits = VT.getScalarSizeInBits();
9739 Tmp += VTBits - PackedOp.getScalarValueSizeInBits();
9740 return Tmp;
9741 }
9742 default:
9743 break;
9744 }
9745 } else {
9746 switch (Opcode) {
9747 case SystemZISD::SELECT_CCMASK:
9748 return computeNumSignBitsBinOp(Op, DemandedElts, DAG, Depth, 0);
9749 default:
9750 break;
9751 }
9752 }
9753
9754 return 1;
9755}
9756
9758 SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG,
9759 UndefPoisonKind Kind, unsigned Depth) const {
9760 switch (Op->getOpcode()) {
9761 case SystemZISD::PCREL_WRAPPER:
9762 case SystemZISD::PCREL_OFFSET:
9763 return true;
9764 }
9765 return false;
9766}
9767
9768unsigned
9770 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
9771 unsigned StackAlign = TFI->getStackAlignment();
9772 assert(StackAlign >=1 && isPowerOf2_32(StackAlign) &&
9773 "Unexpected stack alignment");
9774 // The default stack probe size is 4096 if the function has no
9775 // stack-probe-size attribute.
9776 unsigned StackProbeSize =
9777 MF.getFunction().getFnAttributeAsParsedInteger("stack-probe-size", 4096);
9778 // Round down to the stack alignment.
9779 StackProbeSize &= ~(StackAlign - 1);
9780 return StackProbeSize ? StackProbeSize : StackAlign;
9781}
9782
9783//===----------------------------------------------------------------------===//
9784// Custom insertion
9785//===----------------------------------------------------------------------===//
9786
9787// Force base value Base into a register before MI. Return the register.
9789 const SystemZInstrInfo *TII) {
9790 MachineBasicBlock *MBB = MI.getParent();
9791 MachineFunction &MF = *MBB->getParent();
9792 MachineRegisterInfo &MRI = MF.getRegInfo();
9793
9794 if (Base.isReg()) {
9795 // Copy Base into a new virtual register to help register coalescing in
9796 // cases with multiple uses.
9797 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
9798 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(SystemZ::COPY), Reg)
9799 .add(Base);
9800 return Reg;
9801 }
9802
9803 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
9804 BuildMI(*MBB, MI, MI.getDebugLoc(), TII->get(SystemZ::LA), Reg)
9805 .add(Base)
9806 .addImm(0)
9807 .addReg(0);
9808 return Reg;
9809}
9810
9811// The CC operand of MI might be missing a kill marker because there
9812// were multiple uses of CC, and ISel didn't know which to mark.
9813// Figure out whether MI should have had a kill marker.
9815 // Scan forward through BB for a use/def of CC.
9817 for (MachineBasicBlock::iterator miE = MBB->end(); miI != miE; ++miI) {
9818 const MachineInstr &MI = *miI;
9819 if (MI.readsRegister(SystemZ::CC, /*TRI=*/nullptr))
9820 return false;
9821 if (MI.definesRegister(SystemZ::CC, /*TRI=*/nullptr))
9822 break; // Should have kill-flag - update below.
9823 }
9824
9825 // If we hit the end of the block, check whether CC is live into a
9826 // successor.
9827 if (miI == MBB->end()) {
9828 for (const MachineBasicBlock *Succ : MBB->successors())
9829 if (Succ->isLiveIn(SystemZ::CC))
9830 return false;
9831 }
9832
9833 return true;
9834}
9835
9836// Return true if it is OK for this Select pseudo-opcode to be cascaded
9837// together with other Select pseudo-opcodes into a single basic-block with
9838// a conditional jump around it.
9840 switch (MI.getOpcode()) {
9841 case SystemZ::Select32:
9842 case SystemZ::Select64:
9843 case SystemZ::Select128:
9844 case SystemZ::SelectF32:
9845 case SystemZ::SelectF64:
9846 case SystemZ::SelectF128:
9847 case SystemZ::SelectVR32:
9848 case SystemZ::SelectVR64:
9849 case SystemZ::SelectVR128:
9850 return true;
9851
9852 default:
9853 return false;
9854 }
9855}
9856
9857// Helper function, which inserts PHI functions into SinkMBB:
9858// %Result(i) = phi [ %FalseValue(i), FalseMBB ], [ %TrueValue(i), TrueMBB ],
9859// where %FalseValue(i) and %TrueValue(i) are taken from Selects.
9861 MachineBasicBlock *TrueMBB,
9862 MachineBasicBlock *FalseMBB,
9863 MachineBasicBlock *SinkMBB) {
9864 MachineFunction *MF = TrueMBB->getParent();
9866
9867 MachineInstr *FirstMI = Selects.front();
9868 unsigned CCValid = FirstMI->getOperand(3).getImm();
9869 unsigned CCMask = FirstMI->getOperand(4).getImm();
9870
9871 MachineBasicBlock::iterator SinkInsertionPoint = SinkMBB->begin();
9872
9873 // As we are creating the PHIs, we have to be careful if there is more than
9874 // one. Later Selects may reference the results of earlier Selects, but later
9875 // PHIs have to reference the individual true/false inputs from earlier PHIs.
9876 // That also means that PHI construction must work forward from earlier to
9877 // later, and that the code must maintain a mapping from earlier PHI's
9878 // destination registers, and the registers that went into the PHI.
9880
9881 for (auto *MI : Selects) {
9882 Register DestReg = MI->getOperand(0).getReg();
9883 Register TrueReg = MI->getOperand(1).getReg();
9884 Register FalseReg = MI->getOperand(2).getReg();
9885
9886 // If this Select we are generating is the opposite condition from
9887 // the jump we generated, then we have to swap the operands for the
9888 // PHI that is going to be generated.
9889 if (MI->getOperand(4).getImm() == (CCValid ^ CCMask))
9890 std::swap(TrueReg, FalseReg);
9891
9892 if (auto It = RegRewriteTable.find(TrueReg); It != RegRewriteTable.end())
9893 TrueReg = It->second.first;
9894
9895 if (auto It = RegRewriteTable.find(FalseReg); It != RegRewriteTable.end())
9896 FalseReg = It->second.second;
9897
9898 DebugLoc DL = MI->getDebugLoc();
9899 BuildMI(*SinkMBB, SinkInsertionPoint, DL, TII->get(SystemZ::PHI), DestReg)
9900 .addReg(TrueReg).addMBB(TrueMBB)
9901 .addReg(FalseReg).addMBB(FalseMBB);
9902
9903 // Add this PHI to the rewrite table.
9904 RegRewriteTable[DestReg] = std::make_pair(TrueReg, FalseReg);
9905 }
9906
9907 MF->getProperties().resetNoPHIs();
9908}
9909
9911SystemZTargetLowering::emitAdjCallStack(MachineInstr &MI,
9912 MachineBasicBlock *BB) const {
9913 MachineFunction &MF = *BB->getParent();
9914 MachineFrameInfo &MFI = MF.getFrameInfo();
9915 auto *TFL = Subtarget.getFrameLowering<SystemZFrameLowering>();
9916 assert(TFL->hasReservedCallFrame(MF) &&
9917 "ADJSTACKDOWN and ADJSTACKUP should be no-ops");
9918 (void)TFL;
9919 // Get the MaxCallFrameSize value and erase MI since it serves no further
9920 // purpose as the call frame is statically reserved in the prolog. Set
9921 // AdjustsStack as MI is *not* mapped as a frame instruction.
9922 uint32_t NumBytes = MI.getOperand(0).getImm();
9923 if (NumBytes > MFI.getMaxCallFrameSize())
9924 MFI.setMaxCallFrameSize(NumBytes);
9925 MFI.setAdjustsStack(true);
9926
9927 MI.eraseFromParent();
9928 return BB;
9929}
9930
9931// Implement EmitInstrWithCustomInserter for pseudo Select* instruction MI.
9933SystemZTargetLowering::emitSelect(MachineInstr &MI,
9934 MachineBasicBlock *MBB) const {
9935 assert(isSelectPseudo(MI) && "Bad call to emitSelect()");
9936 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
9937
9938 unsigned CCValid = MI.getOperand(3).getImm();
9939 unsigned CCMask = MI.getOperand(4).getImm();
9940
9941 // If we have a sequence of Select* pseudo instructions using the
9942 // same condition code value, we want to expand all of them into
9943 // a single pair of basic blocks using the same condition.
9944 SmallVector<MachineInstr*, 8> Selects;
9945 SmallVector<MachineInstr*, 8> DbgValues;
9946 Selects.push_back(&MI);
9947 unsigned Count = 0;
9948 for (MachineInstr &NextMI : llvm::make_range(
9949 std::next(MachineBasicBlock::iterator(MI)), MBB->end())) {
9950 if (isSelectPseudo(NextMI)) {
9951 assert(NextMI.getOperand(3).getImm() == CCValid &&
9952 "Bad CCValid operands since CC was not redefined.");
9953 if (NextMI.getOperand(4).getImm() == CCMask ||
9954 NextMI.getOperand(4).getImm() == (CCValid ^ CCMask)) {
9955 Selects.push_back(&NextMI);
9956 continue;
9957 }
9958 break;
9959 }
9960 if (NextMI.definesRegister(SystemZ::CC, /*TRI=*/nullptr) ||
9961 NextMI.usesCustomInsertionHook())
9962 break;
9963 bool User = false;
9964 for (auto *SelMI : Selects)
9965 if (NextMI.readsVirtualRegister(SelMI->getOperand(0).getReg())) {
9966 User = true;
9967 break;
9968 }
9969 if (NextMI.isDebugInstr()) {
9970 if (User) {
9971 assert(NextMI.isDebugValue() && "Unhandled debug opcode.");
9972 DbgValues.push_back(&NextMI);
9973 }
9974 } else if (User || ++Count > 20)
9975 break;
9976 }
9977
9978 MachineInstr *LastMI = Selects.back();
9979 bool CCKilled = (LastMI->killsRegister(SystemZ::CC, /*TRI=*/nullptr) ||
9980 checkCCKill(*LastMI, MBB));
9981 MachineBasicBlock *StartMBB = MBB;
9982 MachineBasicBlock *JoinMBB = SystemZ::splitBlockAfter(LastMI, MBB);
9983 MachineBasicBlock *FalseMBB = SystemZ::emitBlockAfter(StartMBB);
9984
9985 // Unless CC was killed in the last Select instruction, mark it as
9986 // live-in to both FalseMBB and JoinMBB.
9987 if (!CCKilled) {
9988 FalseMBB->addLiveIn(SystemZ::CC);
9989 JoinMBB->addLiveIn(SystemZ::CC);
9990 }
9991
9992 // StartMBB:
9993 // BRC CCMask, JoinMBB
9994 // # fallthrough to FalseMBB
9995 MBB = StartMBB;
9996 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::BRC))
9997 .addImm(CCValid).addImm(CCMask).addMBB(JoinMBB);
9998 MBB->addSuccessor(JoinMBB);
9999 MBB->addSuccessor(FalseMBB);
10000
10001 // FalseMBB:
10002 // # fallthrough to JoinMBB
10003 MBB = FalseMBB;
10004 MBB->addSuccessor(JoinMBB);
10005
10006 // JoinMBB:
10007 // %Result = phi [ %FalseReg, FalseMBB ], [ %TrueReg, StartMBB ]
10008 // ...
10009 MBB = JoinMBB;
10010 createPHIsForSelects(Selects, StartMBB, FalseMBB, MBB);
10011 for (auto *SelMI : Selects)
10012 SelMI->eraseFromParent();
10013
10015 for (auto *DbgMI : DbgValues)
10016 MBB->splice(InsertPos, StartMBB, DbgMI);
10017
10018 return JoinMBB;
10019}
10020
10021// Implement EmitInstrWithCustomInserter for pseudo CondStore* instruction MI.
10022// StoreOpcode is the store to use and Invert says whether the store should
10023// happen when the condition is false rather than true. If a STORE ON
10024// CONDITION is available, STOCOpcode is its opcode, otherwise it is 0.
10025MachineBasicBlock *SystemZTargetLowering::emitCondStore(MachineInstr &MI,
10027 unsigned StoreOpcode,
10028 unsigned STOCOpcode,
10029 bool Invert) const {
10030 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10031
10032 Register SrcReg = MI.getOperand(0).getReg();
10033 MachineOperand Base = MI.getOperand(1);
10034 int64_t Disp = MI.getOperand(2).getImm();
10035 Register IndexReg = MI.getOperand(3).getReg();
10036 unsigned CCValid = MI.getOperand(4).getImm();
10037 unsigned CCMask = MI.getOperand(5).getImm();
10038 DebugLoc DL = MI.getDebugLoc();
10039
10040 StoreOpcode = TII->getOpcodeForOffset(StoreOpcode, Disp);
10041
10042 // ISel pattern matching also adds a load memory operand of the same
10043 // address, so take special care to find the storing memory operand.
10044 MachineMemOperand *MMO = nullptr;
10045 for (auto *I : MI.memoperands())
10046 if (I->isStore()) {
10047 MMO = I;
10048 break;
10049 }
10050
10051 // Use STOCOpcode if possible. We could use different store patterns in
10052 // order to avoid matching the index register, but the performance trade-offs
10053 // might be more complicated in that case.
10054 if (STOCOpcode && !IndexReg && Subtarget.hasLoadStoreOnCond()) {
10055 if (Invert)
10056 CCMask ^= CCValid;
10057
10058 BuildMI(*MBB, MI, DL, TII->get(STOCOpcode))
10059 .addReg(SrcReg)
10060 .add(Base)
10061 .addImm(Disp)
10062 .addImm(CCValid)
10063 .addImm(CCMask)
10064 .addMemOperand(MMO);
10065
10066 MI.eraseFromParent();
10067 return MBB;
10068 }
10069
10070 // Get the condition needed to branch around the store.
10071 if (!Invert)
10072 CCMask ^= CCValid;
10073
10074 MachineBasicBlock *StartMBB = MBB;
10075 MachineBasicBlock *JoinMBB = SystemZ::splitBlockBefore(MI, MBB);
10076 MachineBasicBlock *FalseMBB = SystemZ::emitBlockAfter(StartMBB);
10077
10078 // Unless CC was killed in the CondStore instruction, mark it as
10079 // live-in to both FalseMBB and JoinMBB.
10080 if (!MI.killsRegister(SystemZ::CC, /*TRI=*/nullptr) &&
10081 !checkCCKill(MI, JoinMBB)) {
10082 FalseMBB->addLiveIn(SystemZ::CC);
10083 JoinMBB->addLiveIn(SystemZ::CC);
10084 }
10085
10086 // StartMBB:
10087 // BRC CCMask, JoinMBB
10088 // # fallthrough to FalseMBB
10089 MBB = StartMBB;
10090 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10091 .addImm(CCValid).addImm(CCMask).addMBB(JoinMBB);
10092 MBB->addSuccessor(JoinMBB);
10093 MBB->addSuccessor(FalseMBB);
10094
10095 // FalseMBB:
10096 // store %SrcReg, %Disp(%Index,%Base)
10097 // # fallthrough to JoinMBB
10098 MBB = FalseMBB;
10099 BuildMI(MBB, DL, TII->get(StoreOpcode))
10100 .addReg(SrcReg)
10101 .add(Base)
10102 .addImm(Disp)
10103 .addReg(IndexReg)
10104 .addMemOperand(MMO);
10105 MBB->addSuccessor(JoinMBB);
10106
10107 MI.eraseFromParent();
10108 return JoinMBB;
10109}
10110
10111// Implement EmitInstrWithCustomInserter for pseudo [SU]Cmp128Hi instruction MI.
10113SystemZTargetLowering::emitICmp128Hi(MachineInstr &MI,
10115 bool Unsigned) const {
10116 MachineFunction &MF = *MBB->getParent();
10117 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10118 MachineRegisterInfo &MRI = MF.getRegInfo();
10119
10120 // Synthetic instruction to compare 128-bit values.
10121 // Sets CC 1 if Op0 > Op1, sets a different CC otherwise.
10122 Register Op0 = MI.getOperand(0).getReg();
10123 Register Op1 = MI.getOperand(1).getReg();
10124
10125 MachineBasicBlock *StartMBB = MBB;
10126 MachineBasicBlock *JoinMBB = SystemZ::splitBlockAfter(MI, MBB);
10127 MachineBasicBlock *HiEqMBB = SystemZ::emitBlockAfter(StartMBB);
10128
10129 // StartMBB:
10130 //
10131 // Use VECTOR ELEMENT COMPARE [LOGICAL] to compare the high parts.
10132 // Swap the inputs to get:
10133 // CC 1 if high(Op0) > high(Op1)
10134 // CC 2 if high(Op0) < high(Op1)
10135 // CC 0 if high(Op0) == high(Op1)
10136 //
10137 // If CC != 0, we'd done, so jump over the next instruction.
10138 //
10139 // VEC[L]G Op1, Op0
10140 // JNE JoinMBB
10141 // # fallthrough to HiEqMBB
10142 MBB = StartMBB;
10143 int HiOpcode = Unsigned? SystemZ::VECLG : SystemZ::VECG;
10144 BuildMI(MBB, MI.getDebugLoc(), TII->get(HiOpcode))
10145 .addReg(Op1).addReg(Op0);
10146 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::BRC))
10148 MBB->addSuccessor(JoinMBB);
10149 MBB->addSuccessor(HiEqMBB);
10150
10151 // HiEqMBB:
10152 //
10153 // Otherwise, use VECTOR COMPARE HIGH LOGICAL.
10154 // Since we already know the high parts are equal, the CC
10155 // result will only depend on the low parts:
10156 // CC 1 if low(Op0) > low(Op1)
10157 // CC 3 if low(Op0) <= low(Op1)
10158 //
10159 // VCHLGS Tmp, Op0, Op1
10160 // # fallthrough to JoinMBB
10161 MBB = HiEqMBB;
10162 Register Temp = MRI.createVirtualRegister(&SystemZ::VR128BitRegClass);
10163 BuildMI(MBB, MI.getDebugLoc(), TII->get(SystemZ::VCHLGS), Temp)
10164 .addReg(Op0).addReg(Op1);
10165 MBB->addSuccessor(JoinMBB);
10166
10167 // Mark CC as live-in to JoinMBB.
10168 JoinMBB->addLiveIn(SystemZ::CC);
10169
10170 MI.eraseFromParent();
10171 return JoinMBB;
10172}
10173
10174// Implement EmitInstrWithCustomInserter for subword pseudo ATOMIC_LOADW_* or
10175// ATOMIC_SWAPW instruction MI. BinOpcode is the instruction that performs
10176// the binary operation elided by "*", or 0 for ATOMIC_SWAPW. Invert says
10177// whether the field should be inverted after performing BinOpcode (e.g. for
10178// NAND).
10179MachineBasicBlock *SystemZTargetLowering::emitAtomicLoadBinary(
10180 MachineInstr &MI, MachineBasicBlock *MBB, unsigned BinOpcode,
10181 bool Invert) const {
10182 MachineFunction &MF = *MBB->getParent();
10183 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10184 MachineRegisterInfo &MRI = MF.getRegInfo();
10185
10186 // Extract the operands. Base can be a register or a frame index.
10187 // Src2 can be a register or immediate.
10188 Register Dest = MI.getOperand(0).getReg();
10189 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10190 int64_t Disp = MI.getOperand(2).getImm();
10191 MachineOperand Src2 = earlyUseOperand(MI.getOperand(3));
10192 Register BitShift = MI.getOperand(4).getReg();
10193 Register NegBitShift = MI.getOperand(5).getReg();
10194 unsigned BitSize = MI.getOperand(6).getImm();
10195 DebugLoc DL = MI.getDebugLoc();
10196
10197 // Get the right opcodes for the displacement.
10198 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10199 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10200 assert(LOpcode && CSOpcode && "Displacement out of range");
10201
10202 // Create virtual registers for temporary results.
10203 Register OrigVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10204 Register OldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10205 Register NewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10206 Register RotatedOldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10207 Register RotatedNewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10208
10209 // Insert a basic block for the main loop.
10210 MachineBasicBlock *StartMBB = MBB;
10211 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10212 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10213
10214 // StartMBB:
10215 // ...
10216 // %OrigVal = L Disp(%Base)
10217 // # fall through to LoopMBB
10218 MBB = StartMBB;
10219 BuildMI(MBB, DL, TII->get(LOpcode), OrigVal).add(Base).addImm(Disp).addReg(0);
10220 MBB->addSuccessor(LoopMBB);
10221
10222 // LoopMBB:
10223 // %OldVal = phi [ %OrigVal, StartMBB ], [ %Dest, LoopMBB ]
10224 // %RotatedOldVal = RLL %OldVal, 0(%BitShift)
10225 // %RotatedNewVal = OP %RotatedOldVal, %Src2
10226 // %NewVal = RLL %RotatedNewVal, 0(%NegBitShift)
10227 // %Dest = CS %OldVal, %NewVal, Disp(%Base)
10228 // JNE LoopMBB
10229 // # fall through to DoneMBB
10230 MBB = LoopMBB;
10231 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10232 .addReg(OrigVal).addMBB(StartMBB)
10233 .addReg(Dest).addMBB(LoopMBB);
10234 BuildMI(MBB, DL, TII->get(SystemZ::RLL), RotatedOldVal)
10235 .addReg(OldVal).addReg(BitShift).addImm(0);
10236 if (Invert) {
10237 // Perform the operation normally and then invert every bit of the field.
10238 Register Tmp = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10239 BuildMI(MBB, DL, TII->get(BinOpcode), Tmp)
10240 .addReg(RotatedOldVal)
10241 .add(Src2)
10242 .setOperandDead(3);
10243 // XILF with the upper BitSize bits set.
10244 BuildMI(MBB, DL, TII->get(SystemZ::XILF), RotatedNewVal)
10245 .addReg(Tmp)
10246 .addImm(-1U << (32 - BitSize))
10247 .setOperandDead(3);
10248 } else if (BinOpcode)
10249 // A simply binary operation.
10250 BuildMI(MBB, DL, TII->get(BinOpcode), RotatedNewVal)
10251 .addReg(RotatedOldVal)
10252 .add(Src2)
10253 .setOperandDead(3);
10254 else
10255 // Use RISBG to rotate Src2 into position and use it to replace the
10256 // field in RotatedOldVal.
10257 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RotatedNewVal)
10258 .addReg(RotatedOldVal)
10259 .addReg(Src2.getReg())
10260 .addImm(32)
10261 .addImm(31 + BitSize)
10262 .addImm(32 - BitSize)
10263 .setOperandDead(6);
10264 BuildMI(MBB, DL, TII->get(SystemZ::RLL), NewVal)
10265 .addReg(RotatedNewVal).addReg(NegBitShift).addImm(0);
10266 BuildMI(MBB, DL, TII->get(CSOpcode), Dest)
10267 .addReg(OldVal)
10268 .addReg(NewVal)
10269 .add(Base)
10270 .addImm(Disp);
10271 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10273 MBB->addSuccessor(LoopMBB);
10274 MBB->addSuccessor(DoneMBB);
10275
10276 MI.eraseFromParent();
10277 return DoneMBB;
10278}
10279
10280// Implement EmitInstrWithCustomInserter for subword pseudo
10281// ATOMIC_LOADW_{,U}{MIN,MAX} instruction MI. CompareOpcode is the
10282// instruction that should be used to compare the current field with the
10283// minimum or maximum value. KeepOldMask is the BRC condition-code mask
10284// for when the current field should be kept.
10285MachineBasicBlock *SystemZTargetLowering::emitAtomicLoadMinMax(
10286 MachineInstr &MI, MachineBasicBlock *MBB, unsigned CompareOpcode,
10287 unsigned KeepOldMask) const {
10288 MachineFunction &MF = *MBB->getParent();
10289 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10290 MachineRegisterInfo &MRI = MF.getRegInfo();
10291
10292 // Extract the operands. Base can be a register or a frame index.
10293 Register Dest = MI.getOperand(0).getReg();
10294 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10295 int64_t Disp = MI.getOperand(2).getImm();
10296 Register Src2 = MI.getOperand(3).getReg();
10297 Register BitShift = MI.getOperand(4).getReg();
10298 Register NegBitShift = MI.getOperand(5).getReg();
10299 unsigned BitSize = MI.getOperand(6).getImm();
10300 DebugLoc DL = MI.getDebugLoc();
10301
10302 // Get the right opcodes for the displacement.
10303 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10304 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10305 assert(LOpcode && CSOpcode && "Displacement out of range");
10306
10307 // Create virtual registers for temporary results.
10308 Register OrigVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10309 Register OldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10310 Register NewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10311 Register RotatedOldVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10312 Register RotatedAltVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10313 Register RotatedNewVal = MRI.createVirtualRegister(&SystemZ::GR32BitRegClass);
10314
10315 // Insert 3 basic blocks for the loop.
10316 MachineBasicBlock *StartMBB = MBB;
10317 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10318 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10319 MachineBasicBlock *UseAltMBB = SystemZ::emitBlockAfter(LoopMBB);
10320 MachineBasicBlock *UpdateMBB = SystemZ::emitBlockAfter(UseAltMBB);
10321
10322 // StartMBB:
10323 // ...
10324 // %OrigVal = L Disp(%Base)
10325 // # fall through to LoopMBB
10326 MBB = StartMBB;
10327 BuildMI(MBB, DL, TII->get(LOpcode), OrigVal).add(Base).addImm(Disp).addReg(0);
10328 MBB->addSuccessor(LoopMBB);
10329
10330 // LoopMBB:
10331 // %OldVal = phi [ %OrigVal, StartMBB ], [ %Dest, UpdateMBB ]
10332 // %RotatedOldVal = RLL %OldVal, 0(%BitShift)
10333 // CompareOpcode %RotatedOldVal, %Src2
10334 // BRC KeepOldMask, UpdateMBB
10335 MBB = LoopMBB;
10336 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10337 .addReg(OrigVal).addMBB(StartMBB)
10338 .addReg(Dest).addMBB(UpdateMBB);
10339 BuildMI(MBB, DL, TII->get(SystemZ::RLL), RotatedOldVal)
10340 .addReg(OldVal).addReg(BitShift).addImm(0);
10341 BuildMI(MBB, DL, TII->get(CompareOpcode))
10342 .addReg(RotatedOldVal).addReg(Src2);
10343 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10344 .addImm(SystemZ::CCMASK_ICMP).addImm(KeepOldMask).addMBB(UpdateMBB);
10345 MBB->addSuccessor(UpdateMBB);
10346 MBB->addSuccessor(UseAltMBB);
10347
10348 // UseAltMBB:
10349 // %RotatedAltVal = RISBG %RotatedOldVal, %Src2, 32, 31 + BitSize, 0
10350 // # fall through to UpdateMBB
10351 MBB = UseAltMBB;
10352 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RotatedAltVal)
10353 .addReg(RotatedOldVal)
10354 .addReg(Src2)
10355 .addImm(32)
10356 .addImm(31 + BitSize)
10357 .addImm(0)
10358 .setOperandDead(6);
10359 MBB->addSuccessor(UpdateMBB);
10360
10361 // UpdateMBB:
10362 // %RotatedNewVal = PHI [ %RotatedOldVal, LoopMBB ],
10363 // [ %RotatedAltVal, UseAltMBB ]
10364 // %NewVal = RLL %RotatedNewVal, 0(%NegBitShift)
10365 // %Dest = CS %OldVal, %NewVal, Disp(%Base)
10366 // JNE LoopMBB
10367 // # fall through to DoneMBB
10368 MBB = UpdateMBB;
10369 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RotatedNewVal)
10370 .addReg(RotatedOldVal).addMBB(LoopMBB)
10371 .addReg(RotatedAltVal).addMBB(UseAltMBB);
10372 BuildMI(MBB, DL, TII->get(SystemZ::RLL), NewVal)
10373 .addReg(RotatedNewVal).addReg(NegBitShift).addImm(0);
10374 BuildMI(MBB, DL, TII->get(CSOpcode), Dest)
10375 .addReg(OldVal)
10376 .addReg(NewVal)
10377 .add(Base)
10378 .addImm(Disp);
10379 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10381 MBB->addSuccessor(LoopMBB);
10382 MBB->addSuccessor(DoneMBB);
10383
10384 MI.eraseFromParent();
10385 return DoneMBB;
10386}
10387
10388// Implement EmitInstrWithCustomInserter for subword pseudo ATOMIC_CMP_SWAPW
10389// instruction MI.
10391SystemZTargetLowering::emitAtomicCmpSwapW(MachineInstr &MI,
10392 MachineBasicBlock *MBB) const {
10393 MachineFunction &MF = *MBB->getParent();
10394 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10395 MachineRegisterInfo &MRI = MF.getRegInfo();
10396
10397 // Extract the operands. Base can be a register or a frame index.
10398 Register Dest = MI.getOperand(0).getReg();
10399 MachineOperand Base = earlyUseOperand(MI.getOperand(1));
10400 int64_t Disp = MI.getOperand(2).getImm();
10401 Register CmpVal = MI.getOperand(3).getReg();
10402 Register OrigSwapVal = MI.getOperand(4).getReg();
10403 Register BitShift = MI.getOperand(5).getReg();
10404 Register NegBitShift = MI.getOperand(6).getReg();
10405 int64_t BitSize = MI.getOperand(7).getImm();
10406 DebugLoc DL = MI.getDebugLoc();
10407
10408 const TargetRegisterClass *RC = &SystemZ::GR32BitRegClass;
10409
10410 // Get the right opcodes for the displacement and zero-extension.
10411 unsigned LOpcode = TII->getOpcodeForOffset(SystemZ::L, Disp);
10412 unsigned CSOpcode = TII->getOpcodeForOffset(SystemZ::CS, Disp);
10413 unsigned ZExtOpcode = BitSize == 8 ? SystemZ::LLCR : SystemZ::LLHR;
10414 assert(LOpcode && CSOpcode && "Displacement out of range");
10415
10416 // Create virtual registers for temporary results.
10417 Register OrigOldVal = MRI.createVirtualRegister(RC);
10418 Register OldVal = MRI.createVirtualRegister(RC);
10419 Register SwapVal = MRI.createVirtualRegister(RC);
10420 Register StoreVal = MRI.createVirtualRegister(RC);
10421 Register OldValRot = MRI.createVirtualRegister(RC);
10422 Register RetryOldVal = MRI.createVirtualRegister(RC);
10423 Register RetrySwapVal = MRI.createVirtualRegister(RC);
10424
10425 // Insert 2 basic blocks for the loop.
10426 MachineBasicBlock *StartMBB = MBB;
10427 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10428 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10429 MachineBasicBlock *SetMBB = SystemZ::emitBlockAfter(LoopMBB);
10430
10431 // StartMBB:
10432 // ...
10433 // %OrigOldVal = L Disp(%Base)
10434 // # fall through to LoopMBB
10435 MBB = StartMBB;
10436 BuildMI(MBB, DL, TII->get(LOpcode), OrigOldVal)
10437 .add(Base)
10438 .addImm(Disp)
10439 .addReg(0);
10440 MBB->addSuccessor(LoopMBB);
10441
10442 // LoopMBB:
10443 // %OldVal = phi [ %OrigOldVal, EntryBB ], [ %RetryOldVal, SetMBB ]
10444 // %SwapVal = phi [ %OrigSwapVal, EntryBB ], [ %RetrySwapVal, SetMBB ]
10445 // %OldValRot = RLL %OldVal, BitSize(%BitShift)
10446 // ^^ The low BitSize bits contain the field
10447 // of interest.
10448 // %RetrySwapVal = RISBG32 %SwapVal, %OldValRot, 32, 63-BitSize, 0
10449 // ^^ Replace the upper 32-BitSize bits of the
10450 // swap value with those that we loaded and rotated.
10451 // %Dest = LL[CH] %OldValRot
10452 // CR %Dest, %CmpVal
10453 // JNE DoneMBB
10454 // # Fall through to SetMBB
10455 MBB = LoopMBB;
10456 BuildMI(MBB, DL, TII->get(SystemZ::PHI), OldVal)
10457 .addReg(OrigOldVal).addMBB(StartMBB)
10458 .addReg(RetryOldVal).addMBB(SetMBB);
10459 BuildMI(MBB, DL, TII->get(SystemZ::PHI), SwapVal)
10460 .addReg(OrigSwapVal).addMBB(StartMBB)
10461 .addReg(RetrySwapVal).addMBB(SetMBB);
10462 BuildMI(MBB, DL, TII->get(SystemZ::RLL), OldValRot)
10463 .addReg(OldVal).addReg(BitShift).addImm(BitSize);
10464 BuildMI(MBB, DL, TII->get(SystemZ::RISBG32), RetrySwapVal)
10465 .addReg(SwapVal)
10466 .addReg(OldValRot)
10467 .addImm(32)
10468 .addImm(63 - BitSize)
10469 .addImm(0)
10470 .setOperandDead(6);
10471 BuildMI(MBB, DL, TII->get(ZExtOpcode), Dest)
10472 .addReg(OldValRot);
10473 BuildMI(MBB, DL, TII->get(SystemZ::CR))
10474 .addReg(Dest).addReg(CmpVal);
10475 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10478 MBB->addSuccessor(DoneMBB);
10479 MBB->addSuccessor(SetMBB);
10480
10481 // SetMBB:
10482 // %StoreVal = RLL %RetrySwapVal, -BitSize(%NegBitShift)
10483 // ^^ Rotate the new field to its proper position.
10484 // %RetryOldVal = CS %OldVal, %StoreVal, Disp(%Base)
10485 // JNE LoopMBB
10486 // # fall through to ExitMBB
10487 MBB = SetMBB;
10488 BuildMI(MBB, DL, TII->get(SystemZ::RLL), StoreVal)
10489 .addReg(RetrySwapVal).addReg(NegBitShift).addImm(-BitSize);
10490 BuildMI(MBB, DL, TII->get(CSOpcode), RetryOldVal)
10491 .addReg(OldVal)
10492 .addReg(StoreVal)
10493 .add(Base)
10494 .addImm(Disp);
10495 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10497 MBB->addSuccessor(LoopMBB);
10498 MBB->addSuccessor(DoneMBB);
10499
10500 // If the CC def wasn't dead in the ATOMIC_CMP_SWAPW, mark CC as live-in
10501 // to the block after the loop. At this point, CC may have been defined
10502 // either by the CR in LoopMBB or by the CS in SetMBB.
10503 if (!MI.registerDefIsDead(SystemZ::CC, /*TRI=*/nullptr))
10504 DoneMBB->addLiveIn(SystemZ::CC);
10505
10506 MI.eraseFromParent();
10507 return DoneMBB;
10508}
10509
10510// Emit a move from two GR64s to a GR128.
10512SystemZTargetLowering::emitPair128(MachineInstr &MI,
10513 MachineBasicBlock *MBB) const {
10514 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10515 const DebugLoc &DL = MI.getDebugLoc();
10516
10517 Register Dest = MI.getOperand(0).getReg();
10518 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::REG_SEQUENCE), Dest)
10519 .add(MI.getOperand(1))
10520 .addImm(SystemZ::subreg_h64)
10521 .add(MI.getOperand(2))
10522 .addImm(SystemZ::subreg_l64);
10523 MI.eraseFromParent();
10524 return MBB;
10525}
10526
10527// Emit an extension from a GR64 to a GR128. ClearEven is true
10528// if the high register of the GR128 value must be cleared or false if
10529// it's "don't care".
10530MachineBasicBlock *SystemZTargetLowering::emitExt128(MachineInstr &MI,
10532 bool ClearEven) const {
10533 MachineFunction &MF = *MBB->getParent();
10534 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10535 MachineRegisterInfo &MRI = MF.getRegInfo();
10536 DebugLoc DL = MI.getDebugLoc();
10537
10538 Register Dest = MI.getOperand(0).getReg();
10539 Register Src = MI.getOperand(1).getReg();
10540 Register In128 = MRI.createVirtualRegister(&SystemZ::GR128BitRegClass);
10541
10542 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::IMPLICIT_DEF), In128);
10543 if (ClearEven) {
10544 Register NewIn128 = MRI.createVirtualRegister(&SystemZ::GR128BitRegClass);
10545 Register Zero64 = MRI.createVirtualRegister(&SystemZ::GR64BitRegClass);
10546
10547 BuildMI(*MBB, MI, DL, TII->get(SystemZ::LLILL), Zero64)
10548 .addImm(0);
10549 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::INSERT_SUBREG), NewIn128)
10550 .addReg(In128).addReg(Zero64).addImm(SystemZ::subreg_h64);
10551 In128 = NewIn128;
10552 }
10553 BuildMI(*MBB, MI, DL, TII->get(TargetOpcode::INSERT_SUBREG), Dest)
10554 .addReg(In128).addReg(Src).addImm(SystemZ::subreg_l64);
10555
10556 MI.eraseFromParent();
10557 return MBB;
10558}
10559
10561SystemZTargetLowering::emitMemMemWrapper(MachineInstr &MI,
10563 unsigned Opcode, bool IsMemset) const {
10564 MachineFunction &MF = *MBB->getParent();
10565 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10566 MachineRegisterInfo &MRI = MF.getRegInfo();
10567 DebugLoc DL = MI.getDebugLoc();
10568
10569 MachineOperand DestBase = earlyUseOperand(MI.getOperand(0));
10570 uint64_t DestDisp = MI.getOperand(1).getImm();
10571 MachineOperand SrcBase = MachineOperand::CreateReg(0U, false);
10572 uint64_t SrcDisp;
10573
10574 // Fold the displacement Disp if it is out of range.
10575 auto foldDisplIfNeeded = [&](MachineOperand &Base, uint64_t &Disp) -> void {
10576 if (!isUInt<12>(Disp)) {
10577 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10578 unsigned Opcode = TII->getOpcodeForOffset(SystemZ::LA, Disp);
10579 BuildMI(*MI.getParent(), MI, MI.getDebugLoc(), TII->get(Opcode), Reg)
10580 .add(Base).addImm(Disp).addReg(0);
10582 Disp = 0;
10583 }
10584 };
10585
10586 if (!IsMemset) {
10587 SrcBase = earlyUseOperand(MI.getOperand(2));
10588 SrcDisp = MI.getOperand(3).getImm();
10589 } else {
10590 SrcBase = DestBase;
10591 SrcDisp = DestDisp++;
10592 foldDisplIfNeeded(DestBase, DestDisp);
10593 }
10594
10595 MachineOperand &LengthMO = MI.getOperand(IsMemset ? 2 : 4);
10596 bool IsImmForm = LengthMO.isImm();
10597 bool IsRegForm = !IsImmForm;
10598
10599 // Build and insert one Opcode of Length, with special treatment for memset.
10600 auto insertMemMemOp = [&](MachineBasicBlock *InsMBB,
10602 MachineOperand DBase, uint64_t DDisp,
10603 MachineOperand SBase, uint64_t SDisp,
10604 unsigned Length) -> void {
10605 assert(Length > 0 && Length <= 256 && "Building memory op with bad length.");
10606 if (IsMemset) {
10607 MachineOperand ByteMO = earlyUseOperand(MI.getOperand(3));
10608 if (ByteMO.isImm())
10609 BuildMI(*InsMBB, InsPos, DL, TII->get(SystemZ::MVI))
10610 .add(SBase).addImm(SDisp).add(ByteMO);
10611 else
10612 BuildMI(*InsMBB, InsPos, DL, TII->get(SystemZ::STC))
10613 .add(ByteMO).add(SBase).addImm(SDisp).addReg(0);
10614 if (--Length == 0)
10615 return;
10616 }
10617 BuildMI(*MBB, InsPos, DL, TII->get(Opcode))
10618 .add(DBase).addImm(DDisp).addImm(Length)
10619 .add(SBase).addImm(SDisp)
10620 .setMemRefs(MI.memoperands());
10621 };
10622
10623 bool NeedsLoop = false;
10624 uint64_t ImmLength = 0;
10625 Register LenAdjReg = SystemZ::NoRegister;
10626 if (IsImmForm) {
10627 ImmLength = LengthMO.getImm();
10628 ImmLength += IsMemset ? 2 : 1; // Add back the subtracted adjustment.
10629 if (ImmLength == 0) {
10630 MI.eraseFromParent();
10631 return MBB;
10632 }
10633 if (Opcode == SystemZ::CLC) {
10634 if (ImmLength > 3 * 256)
10635 // A two-CLC sequence is a clear win over a loop, not least because
10636 // it needs only one branch. A three-CLC sequence needs the same
10637 // number of branches as a loop (i.e. 2), but is shorter. That
10638 // brings us to lengths greater than 768 bytes. It seems relatively
10639 // likely that a difference will be found within the first 768 bytes,
10640 // so we just optimize for the smallest number of branch
10641 // instructions, in order to avoid polluting the prediction buffer
10642 // too much.
10643 NeedsLoop = true;
10644 } else if (ImmLength > 6 * 256)
10645 // The heuristic we use is to prefer loops for anything that would
10646 // require 7 or more MVCs. With these kinds of sizes there isn't much
10647 // to choose between straight-line code and looping code, since the
10648 // time will be dominated by the MVCs themselves.
10649 NeedsLoop = true;
10650 } else {
10651 NeedsLoop = true;
10652 LenAdjReg = LengthMO.getReg();
10653 }
10654
10655 // When generating more than one CLC, all but the last will need to
10656 // branch to the end when a difference is found.
10657 MachineBasicBlock *EndMBB =
10658 (Opcode == SystemZ::CLC && (ImmLength > 256 || NeedsLoop)
10660 : nullptr);
10661
10662 if (NeedsLoop) {
10663 Register StartCountReg =
10664 MRI.createVirtualRegister(&SystemZ::GR64BitRegClass);
10665 if (IsImmForm) {
10666 TII->loadImmediate(*MBB, MI, StartCountReg, ImmLength / 256);
10667 ImmLength &= 255;
10668 } else {
10669 BuildMI(*MBB, MI, DL, TII->get(SystemZ::SRLG), StartCountReg)
10670 .addReg(LenAdjReg)
10671 .addReg(0)
10672 .addImm(8);
10673 }
10674
10675 bool HaveSingleBase = DestBase.isIdenticalTo(SrcBase);
10676 auto loadZeroAddress = [&]() -> MachineOperand {
10677 Register Reg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10678 BuildMI(*MBB, MI, DL, TII->get(SystemZ::LGHI), Reg).addImm(0);
10679 return MachineOperand::CreateReg(Reg, false);
10680 };
10681 if (DestBase.isReg() && DestBase.getReg() == SystemZ::NoRegister)
10682 DestBase = loadZeroAddress();
10683 if (SrcBase.isReg() && SrcBase.getReg() == SystemZ::NoRegister)
10684 SrcBase = HaveSingleBase ? DestBase : loadZeroAddress();
10685
10686 MachineBasicBlock *StartMBB = nullptr;
10687 MachineBasicBlock *LoopMBB = nullptr;
10688 MachineBasicBlock *NextMBB = nullptr;
10689 MachineBasicBlock *DoneMBB = nullptr;
10690 MachineBasicBlock *AllDoneMBB = nullptr;
10691
10692 Register StartSrcReg = forceReg(MI, SrcBase, TII);
10693 Register StartDestReg =
10694 (HaveSingleBase ? StartSrcReg : forceReg(MI, DestBase, TII));
10695
10696 const TargetRegisterClass *RC = &SystemZ::ADDR64BitRegClass;
10697 Register ThisSrcReg = MRI.createVirtualRegister(RC);
10698 Register ThisDestReg =
10699 (HaveSingleBase ? ThisSrcReg : MRI.createVirtualRegister(RC));
10700 Register NextSrcReg = MRI.createVirtualRegister(RC);
10701 Register NextDestReg =
10702 (HaveSingleBase ? NextSrcReg : MRI.createVirtualRegister(RC));
10703 RC = &SystemZ::GR64BitRegClass;
10704 Register ThisCountReg = MRI.createVirtualRegister(RC);
10705 Register NextCountReg = MRI.createVirtualRegister(RC);
10706
10707 if (IsRegForm) {
10708 AllDoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10709 StartMBB = SystemZ::emitBlockAfter(MBB);
10710 LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10711 NextMBB = (EndMBB ? SystemZ::emitBlockAfter(LoopMBB) : LoopMBB);
10712 DoneMBB = SystemZ::emitBlockAfter(NextMBB);
10713
10714 // MBB:
10715 // # Jump to AllDoneMBB if LenAdjReg means 0, or fall thru to StartMBB.
10716 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10717 .addReg(LenAdjReg).addImm(IsMemset ? -2 : -1);
10718 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10720 .addMBB(AllDoneMBB);
10721 MBB->addSuccessor(AllDoneMBB);
10722 if (!IsMemset)
10723 MBB->addSuccessor(StartMBB);
10724 else {
10725 // MemsetOneCheckMBB:
10726 // # Jump to MemsetOneMBB for a memset of length 1, or
10727 // # fall thru to StartMBB.
10728 MachineBasicBlock *MemsetOneCheckMBB = SystemZ::emitBlockAfter(MBB);
10729 MachineBasicBlock *MemsetOneMBB = SystemZ::emitBlockAfter(&*MF.rbegin());
10730 MBB->addSuccessor(MemsetOneCheckMBB);
10731 MBB = MemsetOneCheckMBB;
10732 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10733 .addReg(LenAdjReg).addImm(-1);
10734 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10736 .addMBB(MemsetOneMBB);
10737 MBB->addSuccessor(MemsetOneMBB, {10, 100});
10738 MBB->addSuccessor(StartMBB, {90, 100});
10739
10740 // MemsetOneMBB:
10741 // # Jump back to AllDoneMBB after a single MVI or STC.
10742 MBB = MemsetOneMBB;
10743 insertMemMemOp(MBB, MBB->end(),
10744 MachineOperand::CreateReg(StartDestReg, false), DestDisp,
10745 MachineOperand::CreateReg(StartSrcReg, false), SrcDisp,
10746 1);
10747 BuildMI(MBB, DL, TII->get(SystemZ::J)).addMBB(AllDoneMBB);
10748 MBB->addSuccessor(AllDoneMBB);
10749 }
10750
10751 // StartMBB:
10752 // # Jump to DoneMBB if %StartCountReg is zero, or fall through to LoopMBB.
10753 MBB = StartMBB;
10754 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10755 .addReg(StartCountReg).addImm(0);
10756 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10758 .addMBB(DoneMBB);
10759 MBB->addSuccessor(DoneMBB);
10760 MBB->addSuccessor(LoopMBB);
10761 }
10762 else {
10763 StartMBB = MBB;
10764 DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10765 LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10766 NextMBB = (EndMBB ? SystemZ::emitBlockAfter(LoopMBB) : LoopMBB);
10767
10768 // StartMBB:
10769 // # fall through to LoopMBB
10770 MBB->addSuccessor(LoopMBB);
10771
10772 DestBase = MachineOperand::CreateReg(NextDestReg, false);
10773 SrcBase = MachineOperand::CreateReg(NextSrcReg, false);
10774 if (EndMBB && !ImmLength)
10775 // If the loop handled the whole CLC range, DoneMBB will be empty with
10776 // CC live-through into EndMBB, so add it as live-in.
10777 DoneMBB->addLiveIn(SystemZ::CC);
10778 }
10779
10780 // LoopMBB:
10781 // %ThisDestReg = phi [ %StartDestReg, StartMBB ],
10782 // [ %NextDestReg, NextMBB ]
10783 // %ThisSrcReg = phi [ %StartSrcReg, StartMBB ],
10784 // [ %NextSrcReg, NextMBB ]
10785 // %ThisCountReg = phi [ %StartCountReg, StartMBB ],
10786 // [ %NextCountReg, NextMBB ]
10787 // ( PFD 2, 768+DestDisp(%ThisDestReg) )
10788 // Opcode DestDisp(256,%ThisDestReg), SrcDisp(%ThisSrcReg)
10789 // ( JLH EndMBB )
10790 //
10791 // The prefetch is used only for MVC. The JLH is used only for CLC.
10792 MBB = LoopMBB;
10793 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisDestReg)
10794 .addReg(StartDestReg).addMBB(StartMBB)
10795 .addReg(NextDestReg).addMBB(NextMBB);
10796 if (!HaveSingleBase)
10797 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisSrcReg)
10798 .addReg(StartSrcReg).addMBB(StartMBB)
10799 .addReg(NextSrcReg).addMBB(NextMBB);
10800 BuildMI(MBB, DL, TII->get(SystemZ::PHI), ThisCountReg)
10801 .addReg(StartCountReg).addMBB(StartMBB)
10802 .addReg(NextCountReg).addMBB(NextMBB);
10803 if (Opcode == SystemZ::MVC)
10804 BuildMI(MBB, DL, TII->get(SystemZ::PFD))
10806 .addReg(ThisDestReg).addImm(DestDisp - IsMemset + 768).addReg(0);
10807 insertMemMemOp(MBB, MBB->end(),
10808 MachineOperand::CreateReg(ThisDestReg, false), DestDisp,
10809 MachineOperand::CreateReg(ThisSrcReg, false), SrcDisp, 256);
10810 if (EndMBB) {
10811 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10813 .addMBB(EndMBB);
10814 MBB->addSuccessor(EndMBB);
10815 MBB->addSuccessor(NextMBB);
10816 }
10817
10818 // NextMBB:
10819 // %NextDestReg = LA 256(%ThisDestReg)
10820 // %NextSrcReg = LA 256(%ThisSrcReg)
10821 // %NextCountReg = AGHI %ThisCountReg, -1
10822 // CGHI %NextCountReg, 0
10823 // JLH LoopMBB
10824 // # fall through to DoneMBB
10825 //
10826 // The AGHI, CGHI and JLH should be converted to BRCTG by later passes.
10827 MBB = NextMBB;
10828 BuildMI(MBB, DL, TII->get(SystemZ::LA), NextDestReg)
10829 .addReg(ThisDestReg).addImm(256).addReg(0);
10830 if (!HaveSingleBase)
10831 BuildMI(MBB, DL, TII->get(SystemZ::LA), NextSrcReg)
10832 .addReg(ThisSrcReg).addImm(256).addReg(0);
10833 BuildMI(MBB, DL, TII->get(SystemZ::AGHI), NextCountReg)
10834 .addReg(ThisCountReg)
10835 .addImm(-1)
10836 .setOperandDead(3);
10837 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
10838 .addReg(NextCountReg).addImm(0);
10839 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10841 .addMBB(LoopMBB);
10842 MBB->addSuccessor(LoopMBB);
10843 MBB->addSuccessor(DoneMBB);
10844
10845 MBB = DoneMBB;
10846 if (IsRegForm) {
10847 // DoneMBB:
10848 // # Make PHIs for RemDestReg/RemSrcReg as the loop may or may not run.
10849 // # Use EXecute Relative Long for the remainder of the bytes. The target
10850 // instruction of the EXRL will have a length field of 1 since 0 is an
10851 // illegal value. The number of bytes processed becomes (%LenAdjReg &
10852 // 0xff) + 1.
10853 // # Fall through to AllDoneMBB.
10854 Register RemSrcReg = MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10855 Register RemDestReg = HaveSingleBase ? RemSrcReg
10856 : MRI.createVirtualRegister(&SystemZ::ADDR64BitRegClass);
10857 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RemDestReg)
10858 .addReg(StartDestReg).addMBB(StartMBB)
10859 .addReg(NextDestReg).addMBB(NextMBB);
10860 if (!HaveSingleBase)
10861 BuildMI(MBB, DL, TII->get(SystemZ::PHI), RemSrcReg)
10862 .addReg(StartSrcReg).addMBB(StartMBB)
10863 .addReg(NextSrcReg).addMBB(NextMBB);
10864 if (IsMemset)
10865 insertMemMemOp(MBB, MBB->end(),
10866 MachineOperand::CreateReg(RemDestReg, false), DestDisp,
10867 MachineOperand::CreateReg(RemSrcReg, false), SrcDisp, 1);
10868 MachineInstrBuilder EXRL_MIB =
10869 BuildMI(MBB, DL, TII->get(SystemZ::EXRL_Pseudo))
10870 .addImm(Opcode)
10871 .addReg(LenAdjReg)
10872 .addReg(RemDestReg).addImm(DestDisp)
10873 .addReg(RemSrcReg).addImm(SrcDisp);
10874 MBB->addSuccessor(AllDoneMBB);
10875 MBB = AllDoneMBB;
10876 if (Opcode != SystemZ::MVC) {
10877 EXRL_MIB.addReg(SystemZ::CC, RegState::ImplicitDefine);
10878 if (EndMBB)
10879 MBB->addLiveIn(SystemZ::CC);
10880 }
10881 }
10882 MF.getProperties().resetNoPHIs();
10883 }
10884
10885 // Handle any remaining bytes with straight-line code.
10886 while (ImmLength > 0) {
10887 uint64_t ThisLength = std::min(ImmLength, uint64_t(256));
10888 // The previous iteration might have created out-of-range displacements.
10889 // Apply them using LA/LAY if so.
10890 foldDisplIfNeeded(DestBase, DestDisp);
10891 foldDisplIfNeeded(SrcBase, SrcDisp);
10892 insertMemMemOp(MBB, MI, DestBase, DestDisp, SrcBase, SrcDisp, ThisLength);
10893 DestDisp += ThisLength;
10894 SrcDisp += ThisLength;
10895 ImmLength -= ThisLength;
10896 // If there's another CLC to go, branch to the end if a difference
10897 // was found.
10898 if (EndMBB && ImmLength > 0) {
10899 MachineBasicBlock *NextMBB = SystemZ::splitBlockBefore(MI, MBB);
10900 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10902 .addMBB(EndMBB);
10903 MBB->addSuccessor(EndMBB);
10904 MBB->addSuccessor(NextMBB);
10905 MBB = NextMBB;
10906 }
10907 }
10908 if (EndMBB) {
10909 MBB->addSuccessor(EndMBB);
10910 MBB = EndMBB;
10911 MBB->addLiveIn(SystemZ::CC);
10912 }
10913
10914 MI.eraseFromParent();
10915 return MBB;
10916}
10917
10919SystemZTargetLowering::emitMemmoveImm(MachineInstr &MI,
10920 MachineBasicBlock *MBB) const {
10921 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10922
10923 DebugLoc DL = MI.getDebugLoc();
10924 MachineOperand DstAddr = earlyUseOperand(MI.getOperand(0));
10925 MachineOperand SrcAddr = earlyUseOperand(MI.getOperand(1));
10926 uint64_t Len = MI.getOperand(2).getImm();
10927 assert(Len > 0 && Len <= 256 && "Memmove of of unsupported constant length.");
10928
10929 // Use MVC or MVCRL after comparing the addresses.
10930 MachineBasicBlock *DoneMBB = SystemZ::splitBlockAfter(MI, MBB);
10931 MachineBasicBlock *MvcMBB = SystemZ::emitBlockAfter(MBB);
10932 MachineBasicBlock *MvcrlMBB = SystemZ::emitBlockAfter(MvcMBB);
10933 MBB->addSuccessor(MvcMBB);
10934 MBB->addSuccessor(MvcrlMBB);
10935 MvcMBB->addSuccessor(DoneMBB);
10936 MvcrlMBB->addSuccessor(DoneMBB);
10937
10938 BuildMI(MBB, DL, TII->get(SystemZ::CLGR)).add(SrcAddr).add(DstAddr);
10939 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
10941 .addMBB(MvcrlMBB);
10942
10943 BuildMI(MvcMBB, DL, TII->get(SystemZ::MVC))
10944 .add(DstAddr).addImm(0)
10945 .addImm(Len)
10946 .add(SrcAddr).addImm(0)
10947 .setMemRefs(MI.memoperands());
10948 BuildMI(MvcMBB, DL, TII->get(SystemZ::J)).addMBB(DoneMBB);
10949
10950 BuildMI(MvcrlMBB, DL, TII->get(SystemZ::LHI), SystemZ::R0L).addImm(Len - 1);
10951 BuildMI(MvcrlMBB, DL, TII->get(SystemZ::MVCRL))
10952 .add(DstAddr).addImm(0)
10953 .add(SrcAddr).addImm(0)
10954 .setMemRefs(MI.memoperands());
10955
10956 MI.eraseFromParent();
10957 return DoneMBB;
10958}
10959
10960// Decompose string pseudo-instruction MI into a loop that continually performs
10961// Opcode until CC != 3.
10962MachineBasicBlock *SystemZTargetLowering::emitStringWrapper(
10963 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode) const {
10964 MachineFunction &MF = *MBB->getParent();
10965 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
10966 MachineRegisterInfo &MRI = MF.getRegInfo();
10967 DebugLoc DL = MI.getDebugLoc();
10968
10969 uint64_t End1Reg = MI.getOperand(0).getReg();
10970 uint64_t Start1Reg = MI.getOperand(1).getReg();
10971 uint64_t Start2Reg = MI.getOperand(2).getReg();
10972 uint64_t CharReg = MI.getOperand(3).getReg();
10973
10974 const TargetRegisterClass *RC = &SystemZ::GR64BitRegClass;
10975 uint64_t This1Reg = MRI.createVirtualRegister(RC);
10976 uint64_t This2Reg = MRI.createVirtualRegister(RC);
10977 uint64_t End2Reg = MRI.createVirtualRegister(RC);
10978
10979 MachineBasicBlock *StartMBB = MBB;
10980 MachineBasicBlock *DoneMBB = SystemZ::splitBlockBefore(MI, MBB);
10981 MachineBasicBlock *LoopMBB = SystemZ::emitBlockAfter(StartMBB);
10982
10983 // StartMBB:
10984 // # fall through to LoopMBB
10985 MBB->addSuccessor(LoopMBB);
10986
10987 // LoopMBB:
10988 // %This1Reg = phi [ %Start1Reg, StartMBB ], [ %End1Reg, LoopMBB ]
10989 // %This2Reg = phi [ %Start2Reg, StartMBB ], [ %End2Reg, LoopMBB ]
10990 // R0L = %CharReg
10991 // %End1Reg, %End2Reg = CLST %This1Reg, %This2Reg -- uses R0L
10992 // JO LoopMBB
10993 // # fall through to DoneMBB
10994 //
10995 // The load of R0L can be hoisted by post-RA LICM.
10996 MBB = LoopMBB;
10997
10998 BuildMI(MBB, DL, TII->get(SystemZ::PHI), This1Reg)
10999 .addReg(Start1Reg).addMBB(StartMBB)
11000 .addReg(End1Reg).addMBB(LoopMBB);
11001 BuildMI(MBB, DL, TII->get(SystemZ::PHI), This2Reg)
11002 .addReg(Start2Reg).addMBB(StartMBB)
11003 .addReg(End2Reg).addMBB(LoopMBB);
11004 BuildMI(MBB, DL, TII->get(TargetOpcode::COPY), SystemZ::R0L).addReg(CharReg);
11005 BuildMI(MBB, DL, TII->get(Opcode))
11006 .addReg(End1Reg, RegState::Define).addReg(End2Reg, RegState::Define)
11007 .addReg(This1Reg).addReg(This2Reg);
11008 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
11010 MBB->addSuccessor(LoopMBB);
11011 MBB->addSuccessor(DoneMBB);
11012
11013 DoneMBB->addLiveIn(SystemZ::CC);
11014
11015 MI.eraseFromParent();
11016 return DoneMBB;
11017}
11018
11019// Update TBEGIN instruction with final opcode and register clobbers.
11020MachineBasicBlock *SystemZTargetLowering::emitTransactionBegin(
11021 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode,
11022 bool NoFloat) const {
11023 MachineFunction &MF = *MBB->getParent();
11024 const TargetFrameLowering *TFI = Subtarget.getFrameLowering();
11025 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11026
11027 // Update opcode.
11028 MI.setDesc(TII->get(Opcode));
11029
11030 // We cannot handle a TBEGIN that clobbers the stack or frame pointer.
11031 // Make sure to add the corresponding GRSM bits if they are missing.
11032 uint64_t Control = MI.getOperand(2).getImm();
11033 static const unsigned GPRControlBit[16] = {
11034 0x8000, 0x8000, 0x4000, 0x4000, 0x2000, 0x2000, 0x1000, 0x1000,
11035 0x0800, 0x0800, 0x0400, 0x0400, 0x0200, 0x0200, 0x0100, 0x0100
11036 };
11037 Control |= GPRControlBit[15];
11038 if (TFI->hasFP(MF))
11039 Control |= GPRControlBit[11];
11040 MI.getOperand(2).setImm(Control);
11041
11042 // Add GPR clobbers.
11043 for (int I = 0; I < 16; I++) {
11044 if ((Control & GPRControlBit[I]) == 0) {
11045 unsigned Reg = SystemZMC::GR64Regs[I];
11046 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11047 }
11048 }
11049
11050 // Add FPR/VR clobbers.
11051 if (!NoFloat && (Control & 4) != 0) {
11052 if (Subtarget.hasVector()) {
11053 for (unsigned Reg : SystemZMC::VR128Regs) {
11054 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11055 }
11056 } else {
11057 for (unsigned Reg : SystemZMC::FP64Regs) {
11058 MI.addOperand(MachineOperand::CreateReg(Reg, true, true));
11059 }
11060 }
11061 }
11062
11063 return MBB;
11064}
11065
11066MachineBasicBlock *SystemZTargetLowering::emitLoadAndTestCmp0(
11067 MachineInstr &MI, MachineBasicBlock *MBB, unsigned Opcode) const {
11068 MachineFunction &MF = *MBB->getParent();
11069 MachineRegisterInfo *MRI = &MF.getRegInfo();
11070 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11071 DebugLoc DL = MI.getDebugLoc();
11072
11073 Register SrcReg = MI.getOperand(0).getReg();
11074
11075 // Create new virtual register of the same class as source.
11076 const TargetRegisterClass *RC = MRI->getRegClass(SrcReg);
11077 Register DstReg = MRI->createVirtualRegister(RC);
11078
11079 // Replace pseudo with a normal load-and-test that models the def as
11080 // well.
11081 BuildMI(*MBB, MI, DL, TII->get(Opcode), DstReg)
11082 .addReg(SrcReg)
11083 .setMIFlags(MI.getFlags());
11084 MI.eraseFromParent();
11085
11086 return MBB;
11087}
11088
11089MachineBasicBlock *SystemZTargetLowering::emitProbedAlloca(
11091 MachineFunction &MF = *MBB->getParent();
11092 MachineRegisterInfo *MRI = &MF.getRegInfo();
11093 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11094 DebugLoc DL = MI.getDebugLoc();
11095 const unsigned ProbeSize = getStackProbeSize(MF);
11096 Register DstReg = MI.getOperand(0).getReg();
11097 Register SizeReg = MI.getOperand(2).getReg();
11098
11099 MachineBasicBlock *StartMBB = MBB;
11100 MachineBasicBlock *DoneMBB = SystemZ::splitBlockAfter(MI, MBB);
11101 MachineBasicBlock *LoopTestMBB = SystemZ::emitBlockAfter(StartMBB);
11102 MachineBasicBlock *LoopBodyMBB = SystemZ::emitBlockAfter(LoopTestMBB);
11103 MachineBasicBlock *TailTestMBB = SystemZ::emitBlockAfter(LoopBodyMBB);
11104 MachineBasicBlock *TailMBB = SystemZ::emitBlockAfter(TailTestMBB);
11105
11106 MachineMemOperand *VolLdMMO = MF.getMachineMemOperand(MachinePointerInfo(),
11108
11109 Register PHIReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11110 Register IncReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11111
11112 // LoopTestMBB
11113 // BRC TailTestMBB
11114 // # fallthrough to LoopBodyMBB
11115 StartMBB->addSuccessor(LoopTestMBB);
11116 MBB = LoopTestMBB;
11117 BuildMI(MBB, DL, TII->get(SystemZ::PHI), PHIReg)
11118 .addReg(SizeReg)
11119 .addMBB(StartMBB)
11120 .addReg(IncReg)
11121 .addMBB(LoopBodyMBB);
11122 BuildMI(MBB, DL, TII->get(SystemZ::CLGFI))
11123 .addReg(PHIReg)
11124 .addImm(ProbeSize);
11125 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
11127 .addMBB(TailTestMBB);
11128 MBB->addSuccessor(LoopBodyMBB);
11129 MBB->addSuccessor(TailTestMBB);
11130
11131 // LoopBodyMBB: Allocate and probe by means of a volatile compare.
11132 // J LoopTestMBB
11133 MBB = LoopBodyMBB;
11134 BuildMI(MBB, DL, TII->get(SystemZ::SLGFI), IncReg)
11135 .addReg(PHIReg)
11136 .addImm(ProbeSize)
11137 .setOperandDead(3);
11138 BuildMI(MBB, DL, TII->get(SystemZ::SLGFI), SystemZ::R15D)
11139 .addReg(SystemZ::R15D)
11140 .addImm(ProbeSize)
11141 .setOperandDead(3);
11142 BuildMI(MBB, DL, TII->get(SystemZ::CG))
11143 .addReg(SystemZ::R15D)
11144 .addReg(SystemZ::R15D)
11145 .addImm(ProbeSize - 8)
11146 .addReg(0)
11147 .setOperandDead(4)
11148 .setMemRefs(VolLdMMO);
11149 BuildMI(MBB, DL, TII->get(SystemZ::J)).addMBB(LoopTestMBB);
11150 MBB->addSuccessor(LoopTestMBB);
11151
11152 // TailTestMBB
11153 // BRC DoneMBB
11154 // # fallthrough to TailMBB
11155 MBB = TailTestMBB;
11156 BuildMI(MBB, DL, TII->get(SystemZ::CGHI))
11157 .addReg(PHIReg)
11158 .addImm(0);
11159 BuildMI(MBB, DL, TII->get(SystemZ::BRC))
11161 .addMBB(DoneMBB);
11162 MBB->addSuccessor(TailMBB);
11163 MBB->addSuccessor(DoneMBB);
11164
11165 // TailMBB
11166 // # fallthrough to DoneMBB
11167 MBB = TailMBB;
11168 BuildMI(MBB, DL, TII->get(SystemZ::SLGR), SystemZ::R15D)
11169 .addReg(SystemZ::R15D)
11170 .addReg(PHIReg)
11171 .setOperandDead(3);
11172 BuildMI(MBB, DL, TII->get(SystemZ::CG))
11173 .addReg(SystemZ::R15D)
11174 .addReg(SystemZ::R15D)
11175 .addImm(-8)
11176 .addReg(PHIReg)
11177 .setOperandDead(4)
11178 .setMemRefs(VolLdMMO);
11179 MBB->addSuccessor(DoneMBB);
11180
11181 // DoneMBB
11182 MBB = DoneMBB;
11183 BuildMI(*MBB, MBB->begin(), DL, TII->get(TargetOpcode::COPY), DstReg)
11184 .addReg(SystemZ::R15D);
11185
11186 MI.eraseFromParent();
11187 return DoneMBB;
11188}
11189
11190SDValue SystemZTargetLowering::
11191getBackchainAddress(SDValue SP, SelectionDAG &DAG) const {
11193 auto *TFL = Subtarget.getFrameLowering<SystemZELFFrameLowering>();
11194 SDLoc DL(SP);
11195 return DAG.getNode(ISD::ADD, DL, MVT::i64, SP,
11196 DAG.getIntPtrConstant(TFL->getBackchainOffset(MF), DL));
11197}
11198
11199// Replace a _STACKGUARD_DAG pseudo with a _STACKGUARD pseudo, adding
11200// a dead early-clobber def reg that will be used as a scratch register
11201// when the pseudo is expanded.
11202MachineBasicBlock *SystemZTargetLowering::emitStackGuardPseudo(
11203 MachineInstr &MI, MachineBasicBlock *MBB, unsigned PseudoOp) const {
11204 MachineRegisterInfo *MRI = &MBB->getParent()->getRegInfo();
11205 const SystemZInstrInfo *TII = Subtarget.getInstrInfo();
11206 DebugLoc DL = MI.getDebugLoc();
11207 Register AddrReg = MRI->createVirtualRegister(&SystemZ::ADDR64BitRegClass);
11208 BuildMI(*MBB, MI, DL, TII->get(PseudoOp), AddrReg)
11209 .addFrameIndex(MI.getOperand(0).getIndex())
11210 .addImm(MI.getOperand(1).getImm());
11211 MI.eraseFromParent();
11212 return MBB;
11213}
11214
11217 switch (MI.getOpcode()) {
11218 case SystemZ::ADJCALLSTACKDOWN:
11219 case SystemZ::ADJCALLSTACKUP:
11220 return emitAdjCallStack(MI, MBB);
11221
11222 case SystemZ::Select32:
11223 case SystemZ::Select64:
11224 case SystemZ::Select128:
11225 case SystemZ::SelectF32:
11226 case SystemZ::SelectF64:
11227 case SystemZ::SelectF128:
11228 case SystemZ::SelectVR32:
11229 case SystemZ::SelectVR64:
11230 case SystemZ::SelectVR128:
11231 return emitSelect(MI, MBB);
11232
11233 case SystemZ::CondStore8Mux:
11234 return emitCondStore(MI, MBB, SystemZ::STCMux, 0, false);
11235 case SystemZ::CondStore8MuxInv:
11236 return emitCondStore(MI, MBB, SystemZ::STCMux, 0, true);
11237 case SystemZ::CondStore16Mux:
11238 return emitCondStore(MI, MBB, SystemZ::STHMux, 0, false);
11239 case SystemZ::CondStore16MuxInv:
11240 return emitCondStore(MI, MBB, SystemZ::STHMux, 0, true);
11241 case SystemZ::CondStore32Mux:
11242 return emitCondStore(MI, MBB, SystemZ::STMux, SystemZ::STOCMux, false);
11243 case SystemZ::CondStore32MuxInv:
11244 return emitCondStore(MI, MBB, SystemZ::STMux, SystemZ::STOCMux, true);
11245 case SystemZ::CondStore8:
11246 return emitCondStore(MI, MBB, SystemZ::STC, 0, false);
11247 case SystemZ::CondStore8Inv:
11248 return emitCondStore(MI, MBB, SystemZ::STC, 0, true);
11249 case SystemZ::CondStore16:
11250 return emitCondStore(MI, MBB, SystemZ::STH, 0, false);
11251 case SystemZ::CondStore16Inv:
11252 return emitCondStore(MI, MBB, SystemZ::STH, 0, true);
11253 case SystemZ::CondStore32:
11254 return emitCondStore(MI, MBB, SystemZ::ST, SystemZ::STOC, false);
11255 case SystemZ::CondStore32Inv:
11256 return emitCondStore(MI, MBB, SystemZ::ST, SystemZ::STOC, true);
11257 case SystemZ::CondStore64:
11258 return emitCondStore(MI, MBB, SystemZ::STG, SystemZ::STOCG, false);
11259 case SystemZ::CondStore64Inv:
11260 return emitCondStore(MI, MBB, SystemZ::STG, SystemZ::STOCG, true);
11261 case SystemZ::CondStoreF32:
11262 return emitCondStore(MI, MBB, SystemZ::STE, 0, false);
11263 case SystemZ::CondStoreF32Inv:
11264 return emitCondStore(MI, MBB, SystemZ::STE, 0, true);
11265 case SystemZ::CondStoreF64:
11266 return emitCondStore(MI, MBB, SystemZ::STD, 0, false);
11267 case SystemZ::CondStoreF64Inv:
11268 return emitCondStore(MI, MBB, SystemZ::STD, 0, true);
11269
11270 case SystemZ::SCmp128Hi:
11271 return emitICmp128Hi(MI, MBB, false);
11272 case SystemZ::UCmp128Hi:
11273 return emitICmp128Hi(MI, MBB, true);
11274
11275 case SystemZ::PAIR128:
11276 return emitPair128(MI, MBB);
11277 case SystemZ::AEXT128:
11278 return emitExt128(MI, MBB, false);
11279 case SystemZ::ZEXT128:
11280 return emitExt128(MI, MBB, true);
11281
11282 case SystemZ::ATOMIC_SWAPW:
11283 return emitAtomicLoadBinary(MI, MBB, 0);
11284
11285 case SystemZ::ATOMIC_LOADW_AR:
11286 return emitAtomicLoadBinary(MI, MBB, SystemZ::AR);
11287 case SystemZ::ATOMIC_LOADW_AFI:
11288 return emitAtomicLoadBinary(MI, MBB, SystemZ::AFI);
11289
11290 case SystemZ::ATOMIC_LOADW_SR:
11291 return emitAtomicLoadBinary(MI, MBB, SystemZ::SR);
11292
11293 case SystemZ::ATOMIC_LOADW_NR:
11294 return emitAtomicLoadBinary(MI, MBB, SystemZ::NR);
11295 case SystemZ::ATOMIC_LOADW_NILH:
11296 return emitAtomicLoadBinary(MI, MBB, SystemZ::NILH);
11297
11298 case SystemZ::ATOMIC_LOADW_OR:
11299 return emitAtomicLoadBinary(MI, MBB, SystemZ::OR);
11300 case SystemZ::ATOMIC_LOADW_OILH:
11301 return emitAtomicLoadBinary(MI, MBB, SystemZ::OILH);
11302
11303 case SystemZ::ATOMIC_LOADW_XR:
11304 return emitAtomicLoadBinary(MI, MBB, SystemZ::XR);
11305 case SystemZ::ATOMIC_LOADW_XILF:
11306 return emitAtomicLoadBinary(MI, MBB, SystemZ::XILF);
11307
11308 case SystemZ::ATOMIC_LOADW_NRi:
11309 return emitAtomicLoadBinary(MI, MBB, SystemZ::NR, true);
11310 case SystemZ::ATOMIC_LOADW_NILHi:
11311 return emitAtomicLoadBinary(MI, MBB, SystemZ::NILH, true);
11312
11313 case SystemZ::ATOMIC_LOADW_MIN:
11314 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CR, SystemZ::CCMASK_CMP_LE);
11315 case SystemZ::ATOMIC_LOADW_MAX:
11316 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CR, SystemZ::CCMASK_CMP_GE);
11317 case SystemZ::ATOMIC_LOADW_UMIN:
11318 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CLR, SystemZ::CCMASK_CMP_LE);
11319 case SystemZ::ATOMIC_LOADW_UMAX:
11320 return emitAtomicLoadMinMax(MI, MBB, SystemZ::CLR, SystemZ::CCMASK_CMP_GE);
11321
11322 case SystemZ::ATOMIC_CMP_SWAPW:
11323 return emitAtomicCmpSwapW(MI, MBB);
11324 case SystemZ::MVCImm:
11325 case SystemZ::MVCReg:
11326 return emitMemMemWrapper(MI, MBB, SystemZ::MVC);
11327 case SystemZ::NCImm:
11328 return emitMemMemWrapper(MI, MBB, SystemZ::NC);
11329 case SystemZ::OCImm:
11330 return emitMemMemWrapper(MI, MBB, SystemZ::OC);
11331 case SystemZ::XCImm:
11332 case SystemZ::XCReg:
11333 return emitMemMemWrapper(MI, MBB, SystemZ::XC);
11334 case SystemZ::CLCImm:
11335 case SystemZ::CLCReg:
11336 return emitMemMemWrapper(MI, MBB, SystemZ::CLC);
11337 case SystemZ::MemsetImmImm:
11338 case SystemZ::MemsetImmReg:
11339 case SystemZ::MemsetRegImm:
11340 case SystemZ::MemsetRegReg:
11341 return emitMemMemWrapper(MI, MBB, SystemZ::MVC, true/*IsMemset*/);
11342 case SystemZ::MemmoveImm:
11343 return emitMemmoveImm(MI, MBB);
11344 case SystemZ::CLSTLoop:
11345 return emitStringWrapper(MI, MBB, SystemZ::CLST);
11346 case SystemZ::MVSTLoop:
11347 return emitStringWrapper(MI, MBB, SystemZ::MVST);
11348 case SystemZ::SRSTLoop:
11349 return emitStringWrapper(MI, MBB, SystemZ::SRST);
11350 case SystemZ::TBEGIN:
11351 return emitTransactionBegin(MI, MBB, SystemZ::TBEGIN, false);
11352 case SystemZ::TBEGIN_nofloat:
11353 return emitTransactionBegin(MI, MBB, SystemZ::TBEGIN, true);
11354 case SystemZ::TBEGINC:
11355 return emitTransactionBegin(MI, MBB, SystemZ::TBEGINC, true);
11356 case SystemZ::LTEBRCompare_Pseudo:
11357 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTEBR);
11358 case SystemZ::LTDBRCompare_Pseudo:
11359 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTDBR);
11360 case SystemZ::LTXBRCompare_Pseudo:
11361 return emitLoadAndTestCmp0(MI, MBB, SystemZ::LTXBR);
11362
11363 case SystemZ::PROBED_ALLOCA:
11364 return emitProbedAlloca(MI, MBB);
11365 case SystemZ::EH_SjLj_SetJmp:
11366 return emitEHSjLjSetJmp(MI, MBB);
11367 case SystemZ::EH_SjLj_LongJmp:
11368 return emitEHSjLjLongJmp(MI, MBB);
11369
11370 case TargetOpcode::STACKMAP:
11371 case TargetOpcode::PATCHPOINT:
11372 return emitPatchPoint(MI, MBB);
11373
11374 case SystemZ::MOV_STACKGUARD_DAG:
11375 return emitStackGuardPseudo(MI, MBB, SystemZ::MOV_STACKGUARD);
11376
11377 case SystemZ::CMP_STACKGUARD_DAG:
11378 return emitStackGuardPseudo(MI, MBB, SystemZ::CMP_STACKGUARD);
11379
11380 default:
11381 llvm_unreachable("Unexpected instr type to insert");
11382 }
11383}
11384
11385// This is only used by the isel schedulers, and is needed only to prevent
11386// compiler from crashing when list-ilp is used.
11387const TargetRegisterClass *
11388SystemZTargetLowering::getRepRegClassFor(MVT VT) const {
11389 if (VT == MVT::Untyped)
11390 return &SystemZ::ADDR128BitRegClass;
11392}
11393
11394SDValue SystemZTargetLowering::lowerGET_ROUNDING(SDValue Op,
11395 SelectionDAG &DAG) const {
11396 SDLoc dl(Op);
11397 /*
11398 The rounding method is in FPC Byte 3 bits 6-7, and has the following
11399 settings:
11400 00 Round to nearest
11401 01 Round to 0
11402 10 Round to +inf
11403 11 Round to -inf
11404
11405 FLT_ROUNDS, on the other hand, expects the following:
11406 -1 Undefined
11407 0 Round to 0
11408 1 Round to nearest
11409 2 Round to +inf
11410 3 Round to -inf
11411 */
11412
11413 // Save FPC to register.
11414 SDValue Chain = Op.getOperand(0);
11415 SDValue EFPC(
11416 DAG.getMachineNode(SystemZ::EFPC, dl, {MVT::i32, MVT::Other}, Chain), 0);
11417 Chain = EFPC.getValue(1);
11418
11419 // Transform as necessary
11420 SDValue CWD1 = DAG.getNode(ISD::AND, dl, MVT::i32, EFPC,
11421 DAG.getConstant(3, dl, MVT::i32));
11422 // RetVal = (CWD1 ^ (CWD1 >> 1)) ^ 1
11423 SDValue CWD2 = DAG.getNode(ISD::XOR, dl, MVT::i32, CWD1,
11424 DAG.getNode(ISD::SRL, dl, MVT::i32, CWD1,
11425 DAG.getConstant(1, dl, MVT::i32)));
11426
11427 SDValue RetVal = DAG.getNode(ISD::XOR, dl, MVT::i32, CWD2,
11428 DAG.getConstant(1, dl, MVT::i32));
11429 RetVal = DAG.getZExtOrTrunc(RetVal, dl, Op.getValueType());
11430
11431 return DAG.getMergeValues({RetVal, Chain}, dl);
11432}
11433
11434SDValue SystemZTargetLowering::lowerVECREDUCE_ADD(SDValue Op,
11435 SelectionDAG &DAG) const {
11436 EVT VT = Op.getValueType();
11437 Op = Op.getOperand(0);
11438 EVT OpVT = Op.getValueType();
11439
11440 assert(OpVT.isVector() && "Operand type for VECREDUCE_ADD is not a vector.");
11441
11442 SDLoc DL(Op);
11443
11444 // load a 0 vector for the third operand of VSUM.
11445 SDValue Zero = DAG.getSplatBuildVector(OpVT, DL, DAG.getConstant(0, DL, VT));
11446
11447 // execute VSUM.
11448 switch (OpVT.getScalarSizeInBits()) {
11449 case 8:
11450 case 16:
11451 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::v4i32, Op, Zero);
11452 [[fallthrough]];
11453 case 32:
11454 case 64:
11455 Op = DAG.getNode(SystemZISD::VSUM, DL, MVT::i128, Op,
11456 DAG.getBitcast(Op.getValueType(), Zero));
11457 break;
11458 case 128:
11459 break; // VSUM over v1i128 should not happen and would be a noop
11460 default:
11461 llvm_unreachable("Unexpected scalar size.");
11462 }
11463 // Cast to original vector type, retrieve last element.
11464 return DAG.getNode(
11465 ISD::EXTRACT_VECTOR_ELT, DL, VT, DAG.getBitcast(OpVT, Op),
11466 DAG.getConstant(OpVT.getVectorNumElements() - 1, DL, MVT::i32));
11467}
11468
11470 FunctionType *FT = F->getFunctionType();
11471 const AttributeList &Attrs = F->getAttributes();
11472 if (Attrs.hasRetAttrs())
11473 OS << Attrs.getAsString(AttributeList::ReturnIndex) << " ";
11474 OS << *F->getReturnType() << " @" << F->getName() << "(";
11475 for (unsigned I = 0, E = FT->getNumParams(); I != E; ++I) {
11476 if (I)
11477 OS << ", ";
11478 OS << *FT->getParamType(I);
11479 AttributeSet ArgAttrs = Attrs.getParamAttrs(I);
11480 for (auto A : {Attribute::SExt, Attribute::ZExt, Attribute::NoExt})
11481 if (ArgAttrs.hasAttribute(A))
11482 OS << " " << Attribute::getNameFromAttrKind(A);
11483 }
11484 OS << ")\n";
11485}
11486
11487bool SystemZTargetLowering::isInternal(const Function *Fn) const {
11488 std::map<const Function *, bool>::iterator Itr = IsInternalCache.find(Fn);
11489 if (Itr == IsInternalCache.end())
11490 Itr = IsInternalCache
11491 .insert(std::pair<const Function *, bool>(
11492 Fn, (Fn->hasLocalLinkage() && !Fn->hasAddressTaken())))
11493 .first;
11494 return Itr->second;
11495}
11496
11497void SystemZTargetLowering::
11498verifyNarrowIntegerArgs_Call(const SmallVectorImpl<ISD::OutputArg> &Outs,
11499 const Function *F, SDValue Callee) const {
11500 // Temporarily only do the check when explicitly requested, until it can be
11501 // enabled by default.
11503 return;
11504
11505 bool IsInternal = false;
11506 const Function *CalleeFn = nullptr;
11507 if (auto *G = dyn_cast<GlobalAddressSDNode>(Callee))
11508 if ((CalleeFn = dyn_cast<Function>(G->getGlobal())))
11509 IsInternal = isInternal(CalleeFn);
11510 if (!IsInternal && !verifyNarrowIntegerArgs(Outs)) {
11511 errs() << "ERROR: Missing extension attribute of passed "
11512 << "value in call to function:\n" << "Callee: ";
11513 if (CalleeFn != nullptr)
11514 printFunctionArgExts(CalleeFn, errs());
11515 else
11516 errs() << "-\n";
11517 errs() << "Caller: ";
11519 llvm_unreachable("");
11520 }
11521}
11522
11523void SystemZTargetLowering::
11524verifyNarrowIntegerArgs_Ret(const SmallVectorImpl<ISD::OutputArg> &Outs,
11525 const Function *F) const {
11526 // Temporarily only do the check when explicitly requested, until it can be
11527 // enabled by default.
11529 return;
11530
11531 if (!isInternal(F) && !verifyNarrowIntegerArgs(Outs)) {
11532 errs() << "ERROR: Missing extension attribute of returned "
11533 << "value from function:\n";
11535 llvm_unreachable("");
11536 }
11537}
11538
11539// Verify that narrow integer arguments are extended as required by the ABI.
11540// Return false if an error is found.
11541bool SystemZTargetLowering::verifyNarrowIntegerArgs(
11542 const SmallVectorImpl<ISD::OutputArg> &Outs) const {
11543 if (!Subtarget.isTargetELF())
11544 return true;
11545
11548 return true;
11549 } else if (!getTargetMachine().Options.VerifyArgABICompliance)
11550 return true;
11551
11552 for (unsigned i = 0; i < Outs.size(); ++i) {
11553 MVT VT = Outs[i].VT;
11554 ISD::ArgFlagsTy Flags = Outs[i].Flags;
11555 if (VT.isInteger()) {
11556 assert((VT == MVT::i32 || VT.getSizeInBits() >= 64) &&
11557 "Unexpected integer argument VT.");
11558 if (VT == MVT::i32 &&
11559 !Flags.isSExt() && !Flags.isZExt() && !Flags.isNoExt())
11560 return false;
11561 }
11562 }
11563
11564 return true;
11565}
11566
11568 Module &M, const LibcallLoweringInfo &Libcalls) const {
11569 StringRef GuardMode = M.getStackProtectorGuard();
11570
11571 // In the TLS case, no symbol needs to be inserted.
11572 if (GuardMode == "tls" || GuardMode.empty())
11573 return;
11574
11575 // Otherwise (in the global case), insert the appropriate global variable.
11577}
assert(UImm &&(UImm !=~static_cast< T >(0)) &&"Invalid immediate!")
static msgpack::DocNode getNode(msgpack::DocNode DN, msgpack::Type Type, MCValue Val)
unsigned Imm
unsigned uint64_t
AMDGPU Register Bank Select
static bool isZeroVector(SDValue N)
MachineBasicBlock & MBB
MachineBasicBlock MachineBasicBlock::iterator DebugLoc DL
Function Alias Analysis false
Function Alias Analysis Results
#define X(NUM, ENUM, NAME)
Definition ELF.h:857
static GCRegistry::Add< ShadowStackGC > C("shadow-stack", "Very portable GC for uncooperative code generators")
static GCRegistry::Add< ErlangGC > A("erlang", "erlang-compatible garbage collector")
static GCRegistry::Add< CoreCLRGC > E("coreclr", "CoreCLR-compatible GC")
static GCRegistry::Add< OcamlGC > B("ocaml", "ocaml 3.10-compatible GC")
static SDValue convertValVTToLocVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
static SDValue convertLocVTToValVT(SelectionDAG &DAG, SDValue Val, const CCValAssign &VA, const SDLoc &DL)
#define Check(C,...)
const HexagonInstrInfo * TII
IRTranslator LLVM IR MI
Module.h This file contains the declarations for the Module class.
const size_t AbstractManglingParser< Derived, Alloc >::NumOps
const AbstractManglingParser< Derived, Alloc >::OperatorInfo AbstractManglingParser< Derived, Alloc >::Ops[]
#define RegName(no)
static LVOptions Options
Definition LVOptions.cpp:25
static bool isSelectPseudo(MachineInstr &MI)
#define F(x, y, z)
Definition MD5.cpp:54
#define I(x, y, z)
Definition MD5.cpp:57
#define G(x, y, z)
Definition MD5.cpp:55
static bool isUndef(const MachineInstr &MI)
Register Reg
Register const TargetRegisterInfo * TRI
Promote Memory to Register
Definition Mem2Reg.cpp:110
uint64_t High
uint64_t IntrinsicInst * II
#define P(N)
static constexpr MCPhysReg SPReg
const SmallVectorImpl< MachineOperand > & Cond
static cl::opt< RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode > Mode("regalloc-enable-advisor", cl::Hidden, cl::init(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default), cl::desc("Enable regalloc advisor mode"), cl::values(clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Default, "default", "Default"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Release, "release", "precompiled"), clEnumValN(RegAllocEvictionAdvisorAnalysisLegacy::AdvisorMode::Development, "development", "for training")))
const char * Msg
This file defines the SmallSet class.
#define LLVM_DEBUG(...)
Definition Debug.h:119
static SDValue getI128Select(SelectionDAG &DAG, const SDLoc &DL, Comparison C, SDValue TrueOp, SDValue FalseOp)
static SmallVector< SDValue, 4 > simplifyAssumingCCVal(SDValue &Val, SDValue &CC, SelectionDAG &DAG)
static void adjustForTestUnderMask(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static void printFunctionArgExts(const Function *F, raw_fd_ostream &OS)
static void adjustForLTGFR(Comparison &C)
static void adjustSubwordCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static SDValue joinDwords(SelectionDAG &DAG, const SDLoc &DL, SDValue Op0, SDValue Op1)
#define CONV(X)
static cl::opt< bool > EnableIntArgExtCheck("argext-abi-check", cl::init(false), cl::desc("Verify that narrow int args are properly extended per the " "SystemZ ABI."))
static bool isOnlyUsedByStores(SDValue StoredVal, SelectionDAG &DAG)
static void lowerGR128Binary(SelectionDAG &DAG, const SDLoc &DL, EVT VT, unsigned Opcode, SDValue Op0, SDValue Op1, SDValue &Even, SDValue &Odd)
static void adjustForRedundantAnd(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static SDValue lowerAddrSpaceCast(SDValue Op, SelectionDAG &DAG)
static SDValue buildScalarToVector(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SDValue Value)
static SDValue lowerI128ToGR128(SelectionDAG &DAG, SDValue In)
static bool isSimpleShift(SDValue N, unsigned &ShiftVal)
static SDValue mergeHighParts(SelectionDAG &DAG, const SDLoc &DL, unsigned MergedBits, EVT VT, SDValue Op0, SDValue Op1)
static bool isI128MovedToParts(LoadSDNode *LD, SDNode *&LoPart, SDNode *&HiPart)
static bool chooseShuffleOpNos(int *OpNos, unsigned &OpNo0, unsigned &OpNo1)
static uint32_t findZeroVectorIdx(SDValue *Ops, unsigned Num)
static bool isVectorElementSwap(ArrayRef< int > M, EVT VT)
static void getCSAddressAndShifts(SDValue Addr, SelectionDAG &DAG, SDLoc DL, SDValue &AlignedAddr, SDValue &BitShift, SDValue &NegBitShift)
static bool isShlDoublePermute(const SmallVectorImpl< int > &Bytes, unsigned &StartIndex, unsigned &OpNo0, unsigned &OpNo1)
static SDValue getPermuteNode(SelectionDAG &DAG, const SDLoc &DL, const Permute &P, SDValue Op0, SDValue Op1)
static SDNode * emitIntrinsicWithCCAndChain(SelectionDAG &DAG, SDValue Op, unsigned Opcode)
static SDValue getCCResult(SelectionDAG &DAG, SDValue CCReg)
static void adjustForStackGuardCompare(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool isIntrinsicWithCCAndChain(SDValue Op, unsigned &Opcode, unsigned &CCValid)
static void lowerMUL_LOHI32(SelectionDAG &DAG, const SDLoc &DL, unsigned Extend, SDValue Op0, SDValue Op1, SDValue &Hi, SDValue &Lo)
static bool isF128MovedToParts(LoadSDNode *LD, SDNode *&LoPart, SDNode *&HiPart)
static void createPHIsForSelects(SmallVector< MachineInstr *, 8 > &Selects, MachineBasicBlock *TrueMBB, MachineBasicBlock *FalseMBB, MachineBasicBlock *SinkMBB)
static SDValue getGeneralPermuteNode(SelectionDAG &DAG, const SDLoc &DL, SDValue *Ops, const SmallVectorImpl< int > &Bytes)
static unsigned getVectorComparisonOrInvert(ISD::CondCode CC, CmpMode Mode, bool &Invert)
static unsigned CCMaskForCondCode(ISD::CondCode CC)
static void adjustICmpTruncate(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static void adjustForFNeg(Comparison &C)
static bool isScalarToVector(SDValue Op)
static SDValue emitSETCC(SelectionDAG &DAG, const SDLoc &DL, SDValue CCReg, unsigned CCValid, unsigned CCMask)
static bool matchPermute(const SmallVectorImpl< int > &Bytes, const Permute &P, unsigned &OpNo0, unsigned &OpNo1)
static bool isAddCarryChain(SDValue Carry)
static SDValue emitCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static MachineOperand earlyUseOperand(MachineOperand Op)
static bool canUseSiblingCall(const CCState &ArgCCInfo, SmallVectorImpl< CCValAssign > &ArgLocs, SmallVectorImpl< ISD::OutputArg > &Outs)
static bool getzOSCalleeAndADA(SelectionDAG &DAG, SDValue &Callee, SDValue &ADA, SDLoc &DL, SDValue &Chain)
static SDValue convertToF16(SDValue Op, SelectionDAG &DAG)
static bool combineCCMask(SDValue &CCReg, int &CCValid, int &CCMask, SelectionDAG &DAG)
static bool shouldSwapCmpOperands(const Comparison &C)
static bool isNaturalMemoryOperand(SDValue Op, unsigned ICmpType)
static SDValue getADAEntry(SelectionDAG &DAG, SDValue Val, SDLoc DL, unsigned Offset, bool LoadAdr=false)
static SDNode * emitIntrinsicWithCC(SelectionDAG &DAG, SDValue Op, unsigned Opcode)
static void adjustForSubtraction(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool getVPermMask(SDValue ShuffleOp, SmallVectorImpl< int > &Bytes)
static const Permute PermuteForms[]
static bool isI128MovedFromParts(SDValue Val, SDValue &LoPart, SDValue &HiPart)
static std::pair< SDValue, int > findCCUse(const SDValue &Val, unsigned Depth=0)
static bool isSubBorrowChain(SDValue Carry)
static void adjustICmp128(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static bool analyzeArgSplit(const SmallVectorImpl< ArgTy > &Args, SmallVector< CCValAssign, 16 > &ArgLocs, unsigned I, MVT &PartVT, unsigned &NumParts)
static APInt getDemandedSrcElements(SDValue Op, const APInt &DemandedElts, unsigned OpNo)
static SDValue getAbsolute(SelectionDAG &DAG, const SDLoc &DL, SDValue Op, bool IsNegative)
static unsigned computeNumSignBitsBinOp(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth, unsigned OpNo)
static SDValue expandBitCastI128ToF128(SelectionDAG &DAG, SDValue Src, const SDLoc &SL)
static SDValue tryBuildVectorShuffle(SelectionDAG &DAG, BuildVectorSDNode *BVN)
static SDValue convertFromF16(SDValue Op, SDLoc DL, SelectionDAG &DAG)
static unsigned getVectorComparison(ISD::CondCode CC, CmpMode Mode)
static SDValue lowerGR128ToI128(SelectionDAG &DAG, SDValue In)
static SDValue MergeInputChains(SDNode *N1, SDNode *N2)
static SDValue expandBitCastF128ToI128(SelectionDAG &DAG, SDValue Src, const SDLoc &SL)
static unsigned getTestUnderMaskCond(unsigned BitSize, unsigned CCMask, uint64_t Mask, uint64_t CmpVal, unsigned ICmpType)
static bool isIntrinsicWithCC(SDValue Op, unsigned &Opcode, unsigned &CCValid)
static SDValue expandV4F32ToV2F64(SelectionDAG &DAG, int Start, const SDLoc &DL, SDValue Op, SDValue Chain)
static Comparison getCmp(SelectionDAG &DAG, SDValue CmpOp0, SDValue CmpOp1, ISD::CondCode Cond, const SDLoc &DL, SDValue Chain=SDValue(), bool IsSignaling=false)
static bool checkCCKill(MachineInstr &MI, MachineBasicBlock *MBB)
static Register forceReg(MachineInstr &MI, MachineOperand &Base, const SystemZInstrInfo *TII)
static bool is32Bit(EVT VT)
static std::pair< unsigned, const TargetRegisterClass * > parseRegisterNumber(StringRef Constraint, const TargetRegisterClass *RC, const unsigned *Map, unsigned Size)
static unsigned detectEvenOddMultiplyOperand(const SelectionDAG &DAG, const SystemZSubtarget &Subtarget, SDValue &Op)
static bool matchDoublePermute(const SmallVectorImpl< int > &Bytes, const Permute &P, SmallVectorImpl< int > &Transform)
static Comparison getIntrinsicCmp(SelectionDAG &DAG, unsigned Opcode, SDValue Call, unsigned CCValid, uint64_t CC, ISD::CondCode Cond)
static SDValue buildFPVecFromScalars4(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SmallVectorImpl< SDValue > &Elems, unsigned Pos)
static bool isAbsolute(SDValue CmpOp, SDValue Pos, SDValue Neg)
static AddressingMode getLoadStoreAddrMode(bool HasVector, Type *Ty)
static SDValue buildMergeScalars(SelectionDAG &DAG, const SDLoc &DL, EVT VT, SDValue Op0, SDValue Op1)
static void computeKnownBitsBinOp(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth, unsigned OpNo)
static bool getShuffleInput(const SmallVectorImpl< int > &Bytes, unsigned Start, unsigned BytesPerElement, int &Base)
static AddressingMode supportedAddressingMode(Instruction *I, bool HasVector)
static bool isF128MovedFromParts(SDValue Val, SDValue &LoPart, SDValue &HiPart)
static void adjustZeroCmp(SelectionDAG &DAG, const SDLoc &DL, Comparison &C)
static TableGen::Emitter::Opt Y("gen-skeleton-entry", EmitSkeleton, "Generate example skeleton entry")
Value * RHS
Value * LHS
BinaryOperator * Mul
Class for arbitrary precision integers.
Definition APInt.h:78
static APInt getAllOnes(unsigned numBits)
Return an APInt of a specified width with all bits set.
Definition APInt.h:230
LLVM_ABI APInt zext(unsigned width) const
Zero extend to a new width.
Definition APInt.cpp:1057
static APInt getSignMask(unsigned BitWidth)
Get the SignMask for a specific bit width.
Definition APInt.h:225
uint64_t getZExtValue() const
Get zero extended value.
Definition APInt.h:1560
unsigned getActiveBits() const
Compute the number of active bits in the value.
Definition APInt.h:1532
LLVM_ABI APInt trunc(unsigned width) const
Truncate to new width.
Definition APInt.cpp:970
void setBit(unsigned BitPosition)
Set the given bit to 1 whose position is given as "bitPosition".
Definition APInt.h:1350
static APInt getBitsSet(unsigned numBits, unsigned loBit, unsigned hiBit)
Get a value with a block of bits set.
Definition APInt.h:254
unsigned getBitWidth() const
Return the number of bits in the APInt.
Definition APInt.h:1508
bool isSingleWord() const
Determine if this APInt just has one word to store value.
Definition APInt.h:318
LLVM_ABI void insertBits(const APInt &SubBits, unsigned bitPosition)
Insert the bits from a smaller APInt starting at bitPosition.
Definition APInt.cpp:393
bool isSubsetOf(const APInt &RHS) const
This operation checks that all bits set in this APInt are also set in RHS.
Definition APInt.h:1261
void lshrInPlace(unsigned ShiftAmt)
Logical right-shift this APInt by ShiftAmt in place.
Definition APInt.h:860
APInt lshr(unsigned shiftAmt) const
Logical right-shift function.
Definition APInt.h:853
Represent a constant reference to an array (0 or more elements consecutively in memory),...
Definition ArrayRef.h:40
an instruction that atomically reads a memory location, combines it with another value,...
@ Add
*p = old + v
@ Sub
*p = old - v
@ And
*p = old & v
@ Xor
*p = old ^ v
BinOp getOperation() const
This class holds the attributes for a particular argument, parameter, function, or return value.
Definition Attributes.h:410
LLVM_ABI bool hasAttribute(Attribute::AttrKind Kind) const
Return true if the attribute exists in this set.
LLVM_ABI StringRef getValueAsString() const
Return the attribute's value as a string.
static LLVM_ABI StringRef getNameFromAttrKind(Attribute::AttrKind AttrKind)
LLVM Basic Block Representation.
Definition BasicBlock.h:62
A "pseudo-class" with methods for operating on BUILD_VECTORs.
LLVM_ABI bool isConstantSplat(APInt &SplatValue, APInt &SplatUndef, unsigned &SplatBitSize, bool &HasAnyUndefs, unsigned MinSplatBits=0, bool isBigEndian=false) const
Check if this is a constant splat, and if so, find the smallest element size that splats the vector.
LLVM_ABI bool isConstant() const
CCState - This class holds information needed while lowering arguments and return values.
LLVM_ABI void AnalyzeCallResult(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeCallResult - Analyze the return values of a call, incorporating info about the passed values i...
LLVM_ABI bool CheckReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
CheckReturn - Analyze the return values of a function, returning true if the return can be performed ...
LLVM_ABI void AnalyzeReturn(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeReturn - Analyze the returned values of a return, incorporating info about the result values i...
LLVM_ABI void AnalyzeCallOperands(const SmallVectorImpl< ISD::OutputArg > &Outs, CCAssignFn Fn)
AnalyzeCallOperands - Analyze the outgoing arguments to a call, incorporating info about the passed v...
uint64_t getStackSize() const
Returns the size of the currently allocated portion of the stack.
LLVM_ABI void AnalyzeFormalArguments(const SmallVectorImpl< ISD::InputArg > &Ins, CCAssignFn Fn)
AnalyzeFormalArguments - Analyze an array of argument values, incorporating info about the formals in...
CCValAssign - Represent assignment of one arg/retval to a location.
Register getLocReg() const
LocInfo getLocInfo() const
bool needsCustom() const
bool isExtInLoc() const
int64_t getLocMemOffset() const
This class represents a function call, abstracting a target machine's calling convention.
bool isTailCall() const
MachineConstantPoolValue * getMachineCPVal() const
const Constant * getConstVal() const
uint64_t getZExtValue() const
This is an important base class in LLVM.
Definition Constant.h:43
A parsed version of the target data layout string in and methods for querying it.
Definition DataLayout.h:64
LLVM_ABI unsigned getPointerSize(unsigned AS=0) const
The pointer representation size in bytes, rounded up to a whole number of bytes.
A debug info location.
Definition DebugLoc.h:126
iterator find(const_arg_type_t< KeyT > Val)
Definition DenseMap.h:767
iterator end()
Definition DenseMap.h:687
bool hasAddressTaken(const User **=nullptr, bool IgnoreCallbackUses=false, bool IgnoreAssumeLikeCalls=true, bool IngoreLLVMUsed=false, bool IgnoreARCAttachedCall=false, bool IgnoreCastedDirectCall=false) const
hasAddressTaken - returns true if there are any uses of this function other than direct calls or invo...
Definition Function.cpp:944
Attribute getFnAttribute(Attribute::AttrKind Kind) const
Return the attribute for the given attribute kind.
Definition Function.cpp:769
uint64_t getFnAttributeAsParsedInteger(StringRef Kind, uint64_t Default=0) const
For a string attribute Kind, parse attribute as an integer.
Definition Function.cpp:781
CallingConv::ID getCallingConv() const
getCallingConv()/setCallingConv(CC) - These method get and set the calling convention of this functio...
Definition Function.h:273
bool hasFnAttribute(Attribute::AttrKind Kind) const
Return true if the function has the attribute.
Definition Function.cpp:734
LLVM_ABI const GlobalObject * getAliaseeObject() const
Definition Globals.cpp:730
bool hasLocalLinkage() const
bool hasPrivateLinkage() const
bool hasInternalLinkage() const
A wrapper class for inspecting calls to intrinsic functions.
This is an important class for using LLVM in a threaded context.
Definition LLVMContext.h:68
Tracks which library functions to use for a particular subtarget or function.
An instruction for reading from memory.
This class is used to represent ISD::LOAD nodes.
Machine Value Type.
static auto integer_fixedlen_vector_valuetypes()
SimpleValueType SimpleTy
uint64_t getScalarSizeInBits() const
bool isVector() const
Return true if this is a vector value type.
bool isInteger() const
Return true if this is an integer or a vector integer type.
static auto integer_valuetypes()
TypeSize getSizeInBits() const
Returns the size of the specified MVT in bits.
static auto fixedlen_vector_valuetypes()
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
static MVT getVectorVT(MVT VT, unsigned NumElements)
static MVT getIntegerVT(unsigned BitWidth)
static auto fp_valuetypes()
LLVM_ABI void transferSuccessorsAndUpdatePHIs(MachineBasicBlock *FromMBB)
Transfers all the successors, as in transferSuccessors, and update PHI operands in the successor bloc...
LLVM_ABI void addSuccessor(MachineBasicBlock *Succ, BranchProbability Prob=BranchProbability::getUnknown())
Add Succ as a successor of this MachineBasicBlock.
LLVM_ABI iterator getFirstNonPHI()
Returns a pointer to the first instruction in this block that is not a PHINode instruction.
void addLiveIn(MCRegister PhysReg, LaneBitmask LaneMask=LaneBitmask::getAll())
Adds the specified register as a live in.
const MachineFunction * getParent() const
Return the MachineFunction containing this basic block.
void splice(iterator Where, MachineBasicBlock *Other, iterator From)
Take an instruction from MBB 'Other' at the position From, and insert it into this MBB right before '...
MachineInstrBundleIterator< MachineInstr > iterator
void setMachineBlockAddressTaken()
Set this block to indicate that its address is used as something other than the target of a terminato...
The MachineFrameInfo class represents an abstract stack frame until prolog/epilog code is inserted.
void setMaxCallFrameSize(uint64_t S)
LLVM_ABI int CreateFixedObject(uint64_t Size, int64_t SPOffset, bool IsImmutable, bool isAliased=false)
Create a new object at a fixed location on the stack.
void setFrameAddressIsTaken(bool T)
uint64_t getMaxCallFrameSize() const
Return the maximum size of a call frame that must be allocated for an outgoing function call.
void setReturnAddressIsTaken(bool s)
const TargetSubtargetInfo & getSubtarget() const
getSubtarget - Return the subtarget for which this machine code is being compiled.
MachineFrameInfo & getFrameInfo()
getFrameInfo - Return the frame info object for the current function.
void push_back(MachineBasicBlock *MBB)
reverse_iterator rbegin()
MachineRegisterInfo & getRegInfo()
getRegInfo - Return information about the registers currently in use.
const DataLayout & getDataLayout() const
Return the DataLayout attached to the Module associated to this MF.
Function & getFunction()
Return the LLVM function that this machine code represents.
BasicBlockListType::iterator iterator
Ty * getInfo()
getInfo - Keep track of various per-function pieces of information for backends that would like to do...
const MachineFunctionProperties & getProperties() const
Get the function properties.
Register addLiveIn(MCRegister PReg, const TargetRegisterClass *RC)
addLiveIn - Add the specified physical register as a live-in value and create a corresponding virtual...
MachineMemOperand * getMachineMemOperand(MachinePointerInfo PtrInfo, MachineMemOperand::Flags F, LLT MemTy, Align BaseAlignment, const MMOMetadata &Metadata=MMOMetadata(), SyncScope::ID SSID=SyncScope::System, AtomicOrdering Ordering=AtomicOrdering::NotAtomic, AtomicOrdering FailureOrdering=AtomicOrdering::NotAtomic)
getMachineMemOperand - Allocate a new MachineMemOperand.
MachineBasicBlock * CreateMachineBasicBlock(const BasicBlock *BB=nullptr, std::optional< UniqueBBID > BBID=std::nullopt)
CreateMachineInstr - Allocate a new MachineInstr.
void insert(iterator MBBI, MachineBasicBlock *MBB)
const MachineInstrBuilder & setMemRefs(ArrayRef< MachineMemOperand * > MMOs) const
const MachineInstrBuilder & setOperandDead(unsigned OpIdx) const
const MachineInstrBuilder & addReg(Register RegNo, RegState Flags={}, unsigned SubReg=0) const
Add a new virtual register operand.
const MachineInstrBuilder & addImm(int64_t Val) const
Add a new immediate operand.
const MachineInstrBuilder & add(const MachineOperand &MO) const
const MachineInstrBuilder & addFrameIndex(int Idx) const
const MachineInstrBuilder & addRegMask(const uint32_t *Mask) const
const MachineInstrBuilder & addMBB(MachineBasicBlock *MBB, unsigned TargetFlags=0) const
const MachineInstrBuilder & setMIFlags(unsigned Flags) const
const MachineInstrBuilder & addMemOperand(MachineMemOperand *MMO) const
Representation of each machine instruction.
bool killsRegister(Register Reg, const TargetRegisterInfo *TRI) const
Return true if the MachineInstr kills the specified register.
const MachineOperand & getOperand(unsigned i) const
A description of a memory reference used in the backend.
Flags
Flags values. These may be or'd together.
@ MOVolatile
The memory access is volatile.
@ MODereferenceable
The memory access is dereferenceable (i.e., doesn't trap).
@ MOLoad
The memory access reads data.
@ MOInvariant
The memory access always returns the same value (or traps).
@ MOStore
The memory access writes data.
MachineOperand class - Representation of each machine instruction operand.
int64_t getImm() const
bool isReg() const
isReg - Tests if this is a MO_Register operand.
bool isImm() const
isImm - Tests if this is a MO_Immediate operand.
Register getReg() const
getReg - Returns the register number.
LLVM_ABI bool isIdenticalTo(const MachineOperand &Other) const
Returns true if this operand is identical to the specified operand except for liveness related flags ...
static MachineOperand CreateReg(Register Reg, bool isDef, bool isImp=false, bool isKill=false, bool isDead=false, bool isUndef=false, bool isEarlyClobber=false, unsigned SubReg=0, bool isDebug=false, bool isInternalRead=false, bool isRenamable=false)
MachineRegisterInfo - Keep track of information for virtual and physical registers,...
const TargetRegisterClass * getRegClass(Register Reg) const
Return the register class of the specified virtual register.
LLVM_ABI Register createVirtualRegister(const TargetRegisterClass *RegClass, StringRef Name="")
createVirtualRegister - Create and return a new virtual register in the function with the specified r...
void addLiveIn(MCRegister Reg, Register vreg=Register())
addLiveIn - Add the specified register as a live-in.
A Module instance is used to store all the information related to an LLVM module.
Definition Module.h:68
Wrapper class representing virtual and physical registers.
Definition Register.h:20
Wrapper class for IR location info (IR ordering and DebugLoc) to be passed into SDNode creation funct...
Represents one node in the SelectionDAG.
bool isMachineOpcode() const
Test if this node has a post-isel opcode, directly corresponding to a MachineInstr opcode.
unsigned getOpcode() const
Return the SelectionDAG opcode value for this node.
bool hasOneUse() const
Return true if there is exactly one use of this node.
SDNodeFlags getFlags() const
uint64_t getAsZExtVal() const
Helper method returns the zero-extended integer value of a ConstantSDNode.
unsigned getNumValues() const
Return the number of values defined/returned by this operator.
unsigned getNumOperands() const
Return the number of values used by this operation.
unsigned getMachineOpcode() const
This may only be called if isMachineOpcode returns true.
const SDValue & getOperand(unsigned Num) const
bool hasNUsesOfValue(unsigned NUses, unsigned Value) const
Return true if there are exactly NUSES uses of the indicated value.
EVT getValueType(unsigned ResNo) const
Return the type of a specified result.
iterator_range< user_iterator > users()
void setFlags(SDNodeFlags NewFlags)
Represents a use of a SDNode.
Unlike LLVM values, Selection DAG nodes may return multiple values as the result of a computation.
bool isUndef() const
SDNode * getNode() const
get the SDNode which holds the desired result
bool hasOneUse() const
Return true if there is exactly one node using value ResNo of Node, in exactly one operand.
SDValue getValue(unsigned R) const
EVT getValueType() const
Return the ValueType of the referenced return value.
bool isMachineOpcode() const
TypeSize getValueSizeInBits() const
Returns the size of the value in bits.
const SDValue & getOperand(unsigned i) const
const APInt & getConstantOperandAPInt(unsigned i) const
uint64_t getScalarValueSizeInBits() const
unsigned getResNo() const
get the index which selects a specific result in the SDNode
uint64_t getConstantOperandVal(unsigned i) const
MVT getSimpleValueType() const
Return the simple ValueType of the referenced return value.
unsigned getMachineOpcode() const
unsigned getOpcode() const
This is used to represent a portion of an LLVM function in a low-level Data Dependence DAG representa...
SDValue getTargetGlobalAddress(const GlobalValue *GV, const SDLoc &DL, EVT VT, int64_t offset=0, unsigned TargetFlags=0)
SDValue getExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT, unsigned Opcode)
Convert Op, which must be of integer type, to the integer type VT, by either any/sign/zero-extending ...
const TargetSubtargetInfo & getSubtarget() const
SDValue getCopyToReg(SDValue Chain, const SDLoc &dl, Register Reg, SDValue N)
LLVM_ABI SDValue getMergeValues(ArrayRef< SDValue > Ops, const SDLoc &dl)
Create a MERGE_VALUES node from the given operands.
LLVM_ABI SDVTList getVTList(EVT VT)
Return an SDVTList that represents the list of values specified.
LLVM_ABI SDValue getAllOnesConstant(const SDLoc &DL, EVT VT, bool IsTarget=false, bool IsOpaque=false)
LLVM_ABI MachineSDNode * getMachineNode(unsigned Opcode, const SDLoc &dl, EVT VT)
These are used for target selectors to create a new node with specified return type(s),...
LLVM_ABI SDValue getAtomicLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT MemVT, EVT VT, SDValue Chain, SDValue Ptr, MachineMemOperand *MMO)
LLVM_ABI SDValue getConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offs=0, bool isT=false, unsigned TargetFlags=0)
LLVM_ABI bool isConstantIntBuildVectorOrConstantInt(SDValue N, bool AllowOpaques=true) const
Test whether the given value is a constant int or similar node.
LLVM_ABI SDValue UnrollVectorOp(SDNode *N, unsigned ResNE=0)
Utility function used by legalize and lowering to "unroll" a vector operation by splitting out the sc...
LLVM_ABI SDValue getAddrSpaceCast(const SDLoc &dl, EVT VT, SDValue Ptr, unsigned SrcAS, unsigned DestAS, const SDNodeFlags Flags=SDNodeFlags())
Return an AddrSpaceCastSDNode.
LLVM_ABI SDValue getRegister(Register Reg, EVT VT)
SDValue getGLOBAL_OFFSET_TABLE(EVT VT)
Return a GLOBAL_OFFSET_TABLE node. This does not have a useful SDLoc.
LLVM_ABI SDValue getMemIntrinsicNode(unsigned Opcode, const SDLoc &dl, SDVTList VTList, ArrayRef< SDValue > Ops, EVT MemVT, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags Flags=MachineMemOperand::MOLoad|MachineMemOperand::MOStore, LocationSize Size=LocationSize::precise(0), const AAMDNodes &AAInfo=AAMDNodes())
Creates a MemIntrinsicNode that may produce a result and takes a list of operands.
LLVM_ABI SDValue getAtomic(unsigned Opcode, const SDLoc &dl, EVT MemVT, SDValue Chain, SDValue Ptr, SDValue Val, MachineMemOperand *MMO)
Gets a node for an atomic op, produces result (if relevant) and chain and takes 2 operands.
void addNoMergeSiteInfo(const SDNode *Node, bool NoMerge)
Set NoMergeSiteInfo to be associated with Node if NoMerge is true.
LLVM_ABI SDValue getNOT(const SDLoc &DL, SDValue Val, EVT VT)
Create a bitwise NOT operation as (XOR Val, -1).
LLVM_ABI SDValue getMemcpy(SDValue Chain, const SDLoc &dl, SDValue Dst, SDValue Src, SDValue Size, Align DstAlign, Align SrcAlign, bool isVol, bool AlwaysInline, const CallInst *CI, std::optional< bool > OverrideTailCall, MachinePointerInfo DstPtrInfo, MachinePointerInfo SrcPtrInfo, const AAMDNodes &AAInfo=AAMDNodes(), BatchAAResults *BatchAA=nullptr)
const TargetLowering & getTargetLoweringInfo() const
SDValue getTargetJumpTable(int JTI, EVT VT, unsigned TargetFlags=0)
SDValue getUNDEF(EVT VT)
Return an UNDEF node. UNDEF does not have a useful SDLoc.
SDValue getCALLSEQ_END(SDValue Chain, SDValue Op1, SDValue Op2, SDValue InGlue, const SDLoc &DL)
Return a new CALLSEQ_END node, which always must have a glue result (to ensure it's not CSE'd).
SDValue getBuildVector(EVT VT, const SDLoc &DL, ArrayRef< SDValue > Ops)
Return an ISD::BUILD_VECTOR node.
LLVM_ABI bool isSplatValue(SDValue V, const APInt &DemandedElts, APInt &UndefElts, unsigned Depth=0) const
Test whether V has a splatted value for all the demanded elements.
LLVM_ABI SDValue getTruncStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, SDValue Offset, MachinePointerInfo PtrInfo, EVT SVT, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getBitcast(EVT VT, SDValue V)
Return a bitcast using the SDLoc of the value operand, and casting to the provided type.
SDValue getCopyFromReg(SDValue Chain, const SDLoc &dl, Register Reg, EVT VT)
const DataLayout & getDataLayout() const
SDValue getTargetFrameIndex(int FI, EVT VT)
LLVM_ABI SDValue getStore(SDValue Chain, const SDLoc &dl, SDValue Val, SDValue Ptr, MachinePointerInfo PtrInfo, Align Alignment, MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Helper function to build ISD::STORE nodes.
LLVM_ABI SDValue getConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
Create a ConstantSDNode wrapping a constant value.
SDValue getSignedTargetConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI SDValue getExtLoad(ISD::LoadExtType ExtType, const SDLoc &dl, EVT VT, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, EVT MemVT, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
LLVM_ABI SDValue getSignedConstant(int64_t Val, const SDLoc &DL, EVT VT, bool isTarget=false, bool isOpaque=false)
SDValue getSplatVector(EVT VT, const SDLoc &DL, SDValue Op)
SDValue getCALLSEQ_START(SDValue Chain, uint64_t InSize, uint64_t OutSize, const SDLoc &DL)
Return a new CALLSEQ_START node, that starts new call frame, in which InSize bytes are set up inside ...
LLVM_ABI bool SignBitIsZero(SDValue Op, unsigned Depth=0) const
Return true if the sign bit of Op is known to be zero.
LLVM_ABI SDValue getTargetExtractSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand)
A convenience function for creating TargetInstrInfo::EXTRACT_SUBREG nodes.
LLVM_ABI SDValue getSExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either sign-extending or trunca...
LLVM_ABI SDValue getLoad(EVT VT, const SDLoc &dl, SDValue Chain, SDValue Ptr, MachinePointerInfo PtrInfo, MaybeAlign Alignment=MaybeAlign(), MachineMemOperand::Flags MMOFlags=MachineMemOperand::MONone, const MMOMetadata &Metadata=MMOMetadata())
Loads are not normal binary operators: their result type is not determined by their operands,...
LLVM_ABI SDValue getExternalSymbol(const char *Sym, EVT VT)
const TargetMachine & getTarget() const
LLVM_ABI std::pair< SDValue, SDValue > getStrictFPExtendOrRound(SDValue Op, SDValue Chain, const SDLoc &DL, EVT VT)
Convert Op, which must be a STRICT operation of float type, to the float type VT, by either extending...
LLVM_ABI SDValue getIntPtrConstant(uint64_t Val, const SDLoc &DL, bool isTarget=false)
LLVM_ABI SDValue getValueType(EVT)
LLVM_ABI SDValue getNode(unsigned Opcode, const SDLoc &DL, EVT VT, ArrayRef< SDUse > Ops)
Gets or creates the specified node.
LLVM_ABI SDValue getFPExtendOrRound(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of float type, to the float type VT, by either extending or rounding (by tr...
SDValue getTargetConstant(uint64_t Val, const SDLoc &DL, EVT VT, bool isOpaque=false)
LLVM_ABI unsigned ComputeNumSignBits(SDValue Op, unsigned Depth=0) const
Return the number of times the sign bit of the register is replicated into the other bits.
SDValue getTargetBlockAddress(const BlockAddress *BA, EVT VT, int64_t Offset=0, unsigned TargetFlags=0)
LLVM_ABI void ReplaceAllUsesOfValueWith(SDValue From, SDValue To)
Replace any uses of From with To, leaving uses of other values produced by From.getNode() alone.
MachineFunction & getMachineFunction() const
SDValue getSplatBuildVector(EVT VT, const SDLoc &DL, SDValue Op)
Return a splat ISD::BUILD_VECTOR node, consisting of Op splatted to all elements.
LLVM_ABI SDValue getFrameIndex(int FI, EVT VT, bool isTarget=false)
LLVM_ABI KnownBits computeKnownBits(SDValue Op, unsigned Depth=0) const
Determine which bits of Op are known to be either zero or one and return them in Known.
LLVM_ABI SDValue getRegisterMask(const uint32_t *RegMask)
LLVM_ABI SDValue getZExtOrTrunc(SDValue Op, const SDLoc &DL, EVT VT)
Convert Op, which must be of integer type, to the integer type VT, by either zero-extending or trunca...
LLVM_ABI bool MaskedValueIsZero(SDValue Op, const APInt &Mask, unsigned Depth=0) const
Return true if 'Op & Mask' is known to be zero.
SDValue getObjectPtrOffset(const SDLoc &SL, SDValue Ptr, TypeSize Offset)
Create an add instruction with appropriate flags when used for addressing some offset of an object.
LLVMContext * getContext() const
LLVM_ABI SDValue getTargetExternalSymbol(const char *Sym, EVT VT, unsigned TargetFlags=0)
LLVM_ABI SDValue CreateStackTemporary(TypeSize Bytes, Align Alignment)
Create a stack temporary based on the size in bytes and the alignment.
LLVM_ABI SDNode * UpdateNodeOperands(SDNode *N, SDValue Op)
Mutate the specified node in-place to have the specified operands.
SDValue getTargetConstantPool(const Constant *C, EVT VT, MaybeAlign Align=std::nullopt, int Offset=0, unsigned TargetFlags=0)
LLVM_ABI SDValue getTargetInsertSubreg(int SRIdx, const SDLoc &DL, EVT VT, SDValue Operand, SDValue Subreg)
A convenience function for creating TargetInstrInfo::INSERT_SUBREG nodes.
SDValue getEntryNode() const
Return the token chain corresponding to the entry of the function.
LLVM_ABI std::pair< SDValue, SDValue > SplitScalar(const SDValue &N, const SDLoc &DL, const EVT &LoVT, const EVT &HiVT)
Split the scalar node with EXTRACT_ELEMENT using the provided VTs and return the low/high part.
LLVM_ABI SDValue getVectorShuffle(EVT VT, const SDLoc &dl, SDValue N1, SDValue N2, ArrayRef< int > Mask)
Return an ISD::VECTOR_SHUFFLE node.
This SDNode is used to implement the code generator support for the llvm IR shufflevector instruction...
ArrayRef< int > getMask() const
const_iterator begin() const
Definition SmallSet.h:216
std::pair< const_iterator, bool > insert(const T &V)
insert - Insert an element into the set if it isn't already there.
Definition SmallSet.h:184
size_type size() const
Definition SmallSet.h:171
This class consists of common code factored out of the SmallVector class to reduce code duplication b...
reference emplace_back(ArgTypes &&... Args)
void resize(size_type N)
void push_back(const T &Elt)
This is a 'vector' (really, a variable-sized array), optimized for the case when the array is small.
An instruction for storing to memory.
This class is used to represent ISD::STORE nodes.
Represent a constant reference to a string, i.e.
Definition StringRef.h:56
bool getAsInteger(unsigned Radix, T &Result) const
Parse the current string as an integer of the specified radix.
Definition StringRef.h:490
bool starts_with(StringRef Prefix) const
Check if this string starts with the given Prefix.
Definition StringRef.h:258
constexpr bool empty() const
Check if the string is empty.
Definition StringRef.h:141
StringRef slice(size_t Start, size_t End) const
Return a reference to the substring from [Start, End).
Definition StringRef.h:720
constexpr size_t size() const
Get the string size.
Definition StringRef.h:144
iterator end() const
Definition StringRef.h:116
A switch()-like statement whose cases are string literals.
StringSwitch & Case(StringLiteral S, T Value)
A SystemZ-specific class detailing special use registers particular for calling conventions.
static SystemZConstantPoolValue * Create(const GlobalValue *GV, SystemZCP::SystemZCPModifier Modifier)
const SystemZInstrInfo * getInstrInfo() const override
SystemZCallingConventionRegisters * getSpecialRegisters() const
AtomicExpansionKind shouldExpandAtomicRMWInIR(const AtomicRMWInst *RMW) const override
Returns how the IR-level AtomicExpand pass should expand the given AtomicRMW, if at all.
SDValue LowerOperation(SDValue Op, SelectionDAG &DAG) const override
This callback is invoked for operations that are unsupported by the target, which are registered to u...
EVT getOptimalMemOpType(LLVMContext &Context, const MemOp &Op, const AttributeList &FuncAttributes) const override
Returns the target specific optimal type for load and store operations as a result of memset,...
bool hasInlineStackProbe(const MachineFunction &MF) const override
Returns true if stack probing through inline assembly is requested.
MachineBasicBlock * EmitInstrWithCustomInserter(MachineInstr &MI, MachineBasicBlock *BB) const override
This method should be implemented by targets that mark instructions with the 'usesCustomInserter' fla...
MachineBasicBlock * emitEHSjLjSetJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
AtomicExpansionKind shouldCastAtomicLoadInIR(LoadInst *LI) const override
Returns how the given (atomic) load should be cast by the IR-level AtomicExpand pass.
EVT getSetCCResultType(const DataLayout &DL, LLVMContext &, EVT) const override
Return the ValueType of the result of SETCC operations.
bool allowTruncateForTailCall(Type *, Type *) const override
Return true if a truncation from FromTy to ToTy is permitted when deciding whether a call is in tail ...
SDValue LowerAsmOutputForConstraint(SDValue &Chain, SDValue &Flag, const SDLoc &DL, const AsmOperandInfo &Constraint, SelectionDAG &DAG) const override
SDValue LowerReturn(SDValue Chain, CallingConv::ID CallConv, bool IsVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, const SmallVectorImpl< SDValue > &OutVals, const SDLoc &DL, SelectionDAG &DAG) const override
This hook must be implemented to lower outgoing return values, described by the Outs array,...
MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
MachineBasicBlock * emitEHSjLjLongJmp(MachineInstr &MI, MachineBasicBlock *MBB) const
bool CanLowerReturn(CallingConv::ID CallConv, MachineFunction &MF, bool isVarArg, const SmallVectorImpl< ISD::OutputArg > &Outs, LLVMContext &Context, const Type *RetTy) const override
This hook should be implemented to check whether the return values described by the Outs array can fi...
std::pair< SDValue, SDValue > makeExternalCall(SDValue Chain, SelectionDAG &DAG, const char *CalleeName, EVT RetVT, ArrayRef< SDValue > Ops, CallingConv::ID CallConv, bool IsSigned, SDLoc DL, bool DoesNotReturn, bool IsReturnValueUsed) const
void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const override
Insert SSP declaration if global stack protector is used.
bool mayBeEmittedAsTailCall(const CallInst *CI) const override
Return true if the target may be able emit the call instruction as a tail call.
bool splitValueIntoRegisterParts(SelectionDAG &DAG, const SDLoc &DL, SDValue Val, SDValue *Parts, unsigned NumParts, MVT PartVT, std::optional< CallingConv::ID > CC) const override
Target-specific splitting of values into parts that fit a register storing a legal type.
bool isLegalAddressingMode(const DataLayout &DL, const AddrMode &AM, Type *Ty, unsigned AS, Instruction *I=nullptr) const override
Return true if the addressing mode represented by AM is legal for this target, for a load/store of th...
unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const override
Certain targets require unusual breakdowns of certain types.
bool isGuaranteedNotToBeUndefOrPoisonForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, UndefPoisonKind Kind, unsigned Depth) const override
Return true if this function can prove that Op is never poison and, Kind can be used to track poison ...
SystemZTargetLowering(const TargetMachine &TM, const SystemZSubtarget &STI)
bool isFMAFasterThanFMulAndFAdd(const MachineFunction &MF, EVT VT) const override
Return true if an FMA operation is faster than a pair of fmul and fadd instructions.
bool isLegalICmpImmediate(int64_t Imm) const override
Return true if the specified immediate is legal icmp immediate, that is the target has icmp instructi...
std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const override
Given a physical register constraint (e.g.
TargetLowering::ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const override
Examine constraint string and operand type and determine a weight value.
bool allowsMisalignedMemoryAccesses(EVT VT, unsigned AS, Align Alignment, MachineMemOperand::Flags Flags, unsigned *Fast) const override
Determine if the target supports unaligned memory accesses.
const MCPhysReg * getScratchRegisters(CallingConv::ID CC) const override
Returns a 0 terminated array of registers that can be safely used as scratch registers.
TargetLowering::ConstraintType getConstraintType(StringRef Constraint) const override
Given a constraint, return the type of constraint it is for this target.
Register getExceptionSelectorRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception typeid on entry to a la...
bool isFPImmLegal(const APFloat &Imm, EVT VT, bool ForCodeSize) const override
Returns true if the target can instruction select the specified FP immediate natively.
SDValue joinRegisterPartsIntoValue(SelectionDAG &DAG, const SDLoc &DL, const SDValue *Parts, unsigned NumParts, MVT PartVT, EVT ValueVT, std::optional< CallingConv::ID > CC) const override
Target-specific combining of register parts into its original value.
bool isTruncateFree(Type *, Type *) const override
Return true if it's free to truncate a value of type FromTy to type ToTy.
SDValue useLibCall(SelectionDAG &DAG, RTLIB::Libcall LC, MVT VT, SDValue Arg, SDLoc DL, SDValue Chain, bool IsStrict) const
unsigned ComputeNumSignBitsForTargetNode(SDValue Op, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth) const override
Determine the number of bits in the operation that are sign bits.
void LowerOperationWrapper(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked by the type legalizer to legalize nodes with an illegal operand type but leg...
SDValue PerformDAGCombine(SDNode *N, DAGCombinerInfo &DCI) const override
This method will be invoked for all target nodes and for any target-independent nodes that the target...
SDValue LowerCall(CallLoweringInfo &CLI, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower calls into the specified DAG.
bool isLegalAddImmediate(int64_t Imm) const override
Return true if the specified immediate is legal add immediate, that is the target has add instruction...
CondMergingParams getJumpConditionMergingParams(Instruction::BinaryOps Opc, const Value *Lhs, const Value *Rhs, const Function *F) const override
bool findOptimalMemOpLowering(LLVMContext &Context, std::vector< EVT > &MemOps, unsigned Limit, const MemOp &Op, unsigned DstAS, unsigned SrcAS, const AttributeList &FuncAttributes, EVT *LargestVT=nullptr) const override
Determines the optimal series of memory ops to replace the memset / memcpy.
void ReplaceNodeResults(SDNode *N, SmallVectorImpl< SDValue > &Results, SelectionDAG &DAG) const override
This callback is invoked when a node result type is illegal for the target, and the operation was reg...
void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const override
Lower the specified operand into the Ops vector.
Register getExceptionPointerRegister(ExceptionHandling EH, const Constant *PersonalityFn) const override
If a physical register, this returns the register that receives the exception address on entry to an ...
unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const override
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
AtomicExpansionKind shouldCastAtomicStoreInIR(StoreInst *SI) const override
Returns how the given (atomic) store should be cast by the IR-level AtomicExpand pass into.
Register getRegisterByName(const char *RegName, LLT VT, const MachineFunction &MF) const override
Return the register ID of the name passed in.
bool hasAndNot(SDValue Y) const override
Return true if the target has a bitwise and-not operation: X = ~A & B This can be used to simplify se...
SDValue LowerFormalArguments(SDValue Chain, CallingConv::ID CallConv, bool isVarArg, const SmallVectorImpl< ISD::InputArg > &Ins, const SDLoc &DL, SelectionDAG &DAG, SmallVectorImpl< SDValue > &InVals) const override
This hook must be implemented to lower the incoming (formal) arguments, described by the Ins array,...
void computeKnownBitsForTargetNode(const SDValue Op, KnownBits &Known, const APInt &DemandedElts, const SelectionDAG &DAG, unsigned Depth=0) const override
Determine which of the bits specified in Mask are known to be either zero or one and return them in t...
unsigned getStackProbeSize(const MachineFunction &MF) const
XPLINK64 calling convention specific use registers Particular to z/OS when in 64 bit mode.
Information about stack frame layout on the target.
unsigned getStackAlignment() const
getStackAlignment - This method returns the number of bytes to which the stack pointer must be aligne...
bool hasFP(const MachineFunction &MF) const
hasFP - Return true if the specified function should have a dedicated frame pointer register.
TargetInstrInfo - Interface to description of machine instruction set.
void setBooleanVectorContents(BooleanContent Ty)
Specify how the target extends the result of a vector boolean value from a vector of i1 to a wider ty...
void setOperationAction(unsigned Op, MVT VT, LegalizeAction Action)
Indicate that the specified operation does not work with the specified type and indicate what to do a...
EVT getValueType(const DataLayout &DL, Type *Ty, bool AllowUnknown=false) const
Return the EVT corresponding to this LLVM type.
unsigned MaxStoresPerMemcpyOptSize
Likewise for functions with the OptSize attribute.
MachineBasicBlock * emitPatchPoint(MachineInstr &MI, MachineBasicBlock *MBB) const
Replace/modify any TargetFrameIndex operands with a targte-dependent sequence of memory operands that...
virtual const TargetRegisterClass * getRegClassFor(MVT VT, bool isDivergent=false) const
Return the register class that should be used for the specified value type.
const TargetMachine & getTargetMachine() const
virtual unsigned getNumRegistersForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain targets require unusual breakdowns of certain types.
virtual MVT getRegisterTypeForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT) const
Certain combinations of ABIs, Targets and features require that types are legal for some operations a...
virtual void insertSSPDeclarations(Module &M, const LibcallLoweringInfo &Libcalls) const
Inserts necessary declarations for SSP (stack protection) purpose.
void setMaxAtomicSizeInBitsSupported(unsigned SizeInBits)
Set the maximum atomic operation size supported by the backend.
void setAtomicLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Let target indicate that an extending atomic load of the specified type is legal.
virtual unsigned getVectorTypeBreakdownForCallingConv(LLVMContext &Context, CallingConv::ID CC, EVT VT, EVT &IntermediateVT, unsigned &NumIntermediates, MVT &RegisterVT) const
Certain targets such as MIPS require that some types such as vectors are always broken down into scal...
Register getStackPointerRegisterToSaveRestore() const
If a physical register, this specifies the register that llvm.savestack/llvm.restorestack should save...
void setMinFunctionAlignment(Align Alignment)
Set the target's minimum function alignment.
unsigned MaxStoresPerMemsetOptSize
Likewise for functions with the OptSize attribute.
void setBooleanContents(BooleanContent Ty)
Specify how the target extends the result of integer and floating point boolean values from i1 to a w...
unsigned MaxStoresPerMemmove
Specify maximum number of store instructions per memmove call.
void computeRegisterProperties(const TargetRegisterInfo *TRI)
Once all of the register classes are added, this allows us to compute derived properties we expose.
unsigned MaxStoresPerMemmoveOptSize
Likewise for functions with the OptSize attribute.
void addRegisterClass(MVT VT, const TargetRegisterClass *RC)
Add the specified register class as an available regclass for the specified value type.
bool isTypeLegal(EVT VT) const
Return true if the target has native support for the specified value type.
virtual MVT getPointerTy(const DataLayout &DL, uint32_t AS=0) const
Return the pointer type for the given address space, defaults to the pointer type from the data layou...
void setPrefFunctionAlignment(Align Alignment)
Set the target's preferred function alignment.
bool isOperationLegal(unsigned Op, EVT VT) const
Return true if the specified operation is legal on this target.
unsigned MaxStoresPerMemset
Specify maximum number of store instructions per memset call.
void setTruncStoreAction(MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified truncating store does not work with the specified type and indicate what ...
virtual const TargetRegisterClass * getRepRegClassFor(MVT VT) const
Return the 'representative' register class for the specified value type.
void setStackPointerRegisterToSaveRestore(Register R)
If set to a physical register, this specifies the register that llvm.savestack/llvm....
AtomicExpansionKind
Enum that specifies what an atomic load/AtomicRMWInst is expanded to, if at all.
void setTargetDAGCombine(ArrayRef< ISD::NodeType > NTs)
Targets should invoke this method for each target independent node that they want to provide a custom...
void setLoadExtAction(unsigned ExtType, MVT ValVT, MVT MemVT, LegalizeAction Action)
Indicate that the specified load with extension does not work with the specified type and indicate wh...
virtual bool shouldSignExtendTypeInLibCall(Type *Ty, bool IsSigned) const
Returns true if arguments should be sign-extended in lib calls.
std::vector< ArgListEntry > ArgListTy
unsigned MaxStoresPerMemcpy
Specify maximum number of store instructions per memcpy call.
virtual MVT getPointerMemTy(const DataLayout &DL, uint32_t AS=0) const
Return the in-memory pointer type for the given address space, defaults to the pointer type from the ...
void setSchedulingPreference(Sched::Preference Pref)
Specify the target scheduling preference.
LegalizeAction getOperationAction(unsigned Op, EVT VT) const
Return how this operation should be treated: either it is legal, needs to be promoted to a larger siz...
virtual ConstraintType getConstraintType(StringRef Constraint) const
Given a constraint, return the type of constraint it is for this target.
virtual bool findOptimalMemOpLowering(LLVMContext &Context, std::vector< EVT > &MemOps, unsigned Limit, const MemOp &Op, unsigned DstAS, unsigned SrcAS, const AttributeList &FuncAttributes, EVT *LargestVT=nullptr) const
Determines the optimal series of memory ops to replace the memset / memcpy.
virtual SDValue LowerToTLSEmulatedModel(const GlobalAddressSDNode *GA, SelectionDAG &DAG) const
Lower TLS global address SDNode for target independent emulated TLS model.
std::pair< SDValue, SDValue > LowerCallTo(CallLoweringInfo &CLI) const
This function lowers an abstract call to a function into an actual call.
virtual ConstraintWeight getSingleConstraintMatchWeight(AsmOperandInfo &info, const char *constraint) const
Examine constraint string and operand type and determine a weight value.
virtual std::pair< unsigned, const TargetRegisterClass * > getRegForInlineAsmConstraint(const TargetRegisterInfo *TRI, StringRef Constraint, MVT VT) const
Given a physical register constraint (e.g.
TargetLowering(const TargetLowering &)=delete
virtual void LowerAsmOperandForConstraint(SDValue Op, StringRef Constraint, std::vector< SDValue > &Ops, SelectionDAG &DAG) const
Lower the specified operand into the Ops vector.
std::pair< SDValue, SDValue > makeLibCall(SelectionDAG &DAG, RTLIB::LibcallImpl LibcallImpl, EVT RetVT, ArrayRef< SDValue > Ops, MakeLibCallOptions CallOptions, const SDLoc &dl, SDValue Chain=SDValue()) const
Returns a pair of (return value, chain).
Primary interface to the complete machine description for the target machine.
TLSModel::Model getTLSModel(const GlobalValue *GV) const
Returns the TLS model which should be used for the given global variable.
bool useEmulatedTLS() const
Returns true if this target uses emulated TLS.
CodeModel::Model getCodeModel() const
Returns the code model.
TargetRegisterInfo base class - We assume that the target defines a static array of TargetRegisterDes...
virtual const TargetInstrInfo * getInstrInfo() const
static constexpr TypeSize getFixed(ScalarTy ExactSize)
Definition TypeSize.h:339
The instances of the Type class are immutable: once they are created, they are never changed.
Definition Type.h:46
bool isVectorTy() const
True if this is an instance of VectorType.
Definition Type.h:283
LLVM_ABI TypeSize getPrimitiveSizeInBits() const LLVM_READONLY
Return the basic size of this type if it is a primitive type.
Definition Type.cpp:187
LLVM_ABI unsigned getScalarSizeInBits() const LLVM_READONLY
If this is a vector type, return the getPrimitiveSizeInBits value for the element type.
Definition Type.cpp:222
bool isFloatingPointTy() const
Return true if this is one of the floating-point types.
Definition Type.h:186
bool isIntegerTy() const
True if this is an instance of IntegerType.
Definition Type.h:252
A Use represents the edge between a Value definition and its users.
Definition Use.h:35
User * getUser() const
Returns the User that contains this Use.
Definition Use.h:61
Value * getOperand(unsigned i) const
Definition User.h:207
LLVM Value Representation.
Definition Value.h:75
Type * getType() const
All values are typed, get the type of this value.
Definition Value.h:257
user_iterator user_begin()
Definition Value.h:404
bool hasOneUse() const
Return true if there is exactly one use of this value.
Definition Value.h:441
int getNumOccurrences() const
constexpr ScalarTy getFixedValue() const
Definition TypeSize.h:200
A raw_ostream that writes to a file descriptor.
CallInst * Call
#define llvm_unreachable(msg)
Marks that the current location is not supposed to be reachable.
constexpr char Align[]
Key for Kernel::Arg::Metadata::mAlign.
constexpr std::underlying_type_t< E > Mask()
Get a bitmask with 1s in all places up to the high-order bit of E's largest value.
unsigned ID
LLVM IR allows to use arbitrary numbers as calling convention identifiers.
Definition CallingConv.h:24
@ GHC
Used by the Glasgow Haskell Compiler (GHC).
Definition CallingConv.h:50
@ C
The default llvm calling convention, compatible with C.
Definition CallingConv.h:34
bool isNON_EXTLoad(const SDNode *N)
Returns true if the specified node is a non-extending load.
@ SETCC
SetCC operator - This evaluates to a true value iff the condition is true.
Definition ISDOpcodes.h:837
@ MERGE_VALUES
MERGE_VALUES - This node takes multiple discrete operands and returns them all as its individual resu...
Definition ISDOpcodes.h:263
@ STACKRESTORE
STACKRESTORE has two operands, an input chain and a pointer to restore to it returns an output chain.
@ STACKSAVE
STACKSAVE - STACKSAVE has one operand, an input chain.
@ STRICT_FSETCC
STRICT_FSETCC/STRICT_FSETCCS - Constrained versions of SETCC, used for floating-point operands only.
Definition ISDOpcodes.h:516
@ POISON
POISON - A poison node.
Definition ISDOpcodes.h:238
@ EH_SJLJ_LONGJMP
OUTCHAIN = EH_SJLJ_LONGJMP(INCHAIN, buffer) This corresponds to the eh.sjlj.longjmp intrinsic.
Definition ISDOpcodes.h:170
@ SMUL_LOHI
SMUL_LOHI/UMUL_LOHI - Multiply two integers of type iN, producing a signed/unsigned value of type i[2...
Definition ISDOpcodes.h:277
@ BSWAP
Byte Swap and Counting operators.
Definition ISDOpcodes.h:797
@ VAEND
VAEND, VASTART - VAEND and VASTART have three operands: an input chain, pointer, and a SRCVALUE.
@ ATOMIC_STORE
OUTCHAIN = ATOMIC_STORE(INCHAIN, val, ptr) This corresponds to "store atomic" instruction.
@ ADD
Simple integer binary arithmetic operators.
Definition ISDOpcodes.h:266
@ LOAD
LOAD and STORE have token chains as their first operand, then the same operands as an LLVM load/store...
@ ANY_EXTEND
ANY_EXTEND - Used for integer types. The high bits are undefined.
Definition ISDOpcodes.h:871
@ FMA
FMA - Perform a * b + c with no intermediate rounding step.
Definition ISDOpcodes.h:523
@ PSEUDO_FMIN
PSEUDO_FMIN is strictly equivalent to op0 olt op1 ?
@ INTRINSIC_VOID
OUTCHAIN = INTRINSIC_VOID(INCHAIN, INTRINSICID, arg1, arg2, ...) This node represents a target intrin...
Definition ISDOpcodes.h:222
@ GlobalAddress
Definition ISDOpcodes.h:90
@ ATOMIC_CMP_SWAP_WITH_SUCCESS
Val, Success, OUTCHAIN = ATOMIC_CMP_SWAP_WITH_SUCCESS(INCHAIN, ptr, cmp, swap) N.b.
@ STRICT_FMINIMUM
Definition ISDOpcodes.h:476
@ SINT_TO_FP
[SU]INT_TO_FP - These operators convert integers (whose interpreted sign depends on the first letter)...
Definition ISDOpcodes.h:898
@ FADD
Simple binary floating point operators.
Definition ISDOpcodes.h:420
@ ABS
ABS - Determine the unsigned absolute value of a signed integer value of the same bitwidth.
Definition ISDOpcodes.h:757
@ MEMBARRIER
MEMBARRIER - Compiler barrier only; generate a no-op.
@ ATOMIC_FENCE
OUTCHAIN = ATOMIC_FENCE(INCHAIN, ordering, scope) This corresponds to the fence instruction.
@ SIGN_EXTEND_VECTOR_INREG
SIGN_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register sign-extension of the low ...
Definition ISDOpcodes.h:928
@ SDIVREM
SDIVREM/UDIVREM - Divide two integers and produce both a quotient and remainder result.
Definition ISDOpcodes.h:282
@ BITCAST
BITCAST - This operator converts between integer, vector and FP values, as if the value was stored to...
@ STRICT_PSEUDO_FMAX
Definition ISDOpcodes.h:465
@ BUILD_PAIR
BUILD_PAIR - This is the opposite of EXTRACT_ELEMENT in some ways.
Definition ISDOpcodes.h:256
@ STRICT_FSQRT
Constrained versions of libm-equivalent floating point intrinsics.
Definition ISDOpcodes.h:441
@ BUILTIN_OP_END
BUILTIN_OP_END - This must be the last enum value in this list.
@ GlobalTLSAddress
Definition ISDOpcodes.h:91
@ CTLZ_ZERO_POISON
Definition ISDOpcodes.h:806
@ SIGN_EXTEND
Conversion operators.
Definition ISDOpcodes.h:862
@ STRICT_UINT_TO_FP
Definition ISDOpcodes.h:490
@ SCALAR_TO_VECTOR
SCALAR_TO_VECTOR(VAL) - This represents the operation of loading a scalar value into element 0 of the...
Definition ISDOpcodes.h:675
@ PREFETCH
PREFETCH - This corresponds to a prefetch intrinsic.
@ STRICT_PSEUDO_FMIN
Definition ISDOpcodes.h:464
@ FSINCOS
FSINCOS - Compute both fsin and fcos as a single operation.
@ FNEG
Perform various unary floating-point operations inspired by libm.
@ BR_CC
BR_CC - Conditional branch.
@ SSUBO
Same for subtraction.
Definition ISDOpcodes.h:355
@ BR_JT
BR_JT - Jumptable branch.
@ IS_FPCLASS
Performs a check of floating point class property, defined by IEEE-754.
Definition ISDOpcodes.h:553
@ SELECT
Select(COND, TRUEVAL, FALSEVAL).
Definition ISDOpcodes.h:814
@ ATOMIC_LOAD
Val, OUTCHAIN = ATOMIC_LOAD(INCHAIN, ptr) This corresponds to "load atomic" instruction.
@ UNDEF
UNDEF - An undefined node.
Definition ISDOpcodes.h:235
@ EXTRACT_ELEMENT
EXTRACT_ELEMENT - This is used to get the lower or upper (determined by a Constant,...
Definition ISDOpcodes.h:249
@ SPLAT_VECTOR
SPLAT_VECTOR(VAL) - Returns a vector with the scalar value VAL duplicated in all lanes.
Definition ISDOpcodes.h:682
@ VACOPY
VACOPY - VACOPY has 5 operands: an input chain, a destination pointer, a source pointer,...
@ SADDO
RESULT, BOOL = [SU]ADDO(LHS, RHS) - Overflow-aware nodes for addition.
Definition ISDOpcodes.h:351
@ VECREDUCE_ADD
Integer reductions may have a result type larger than the vector element type.
@ GET_ROUNDING
Returns current rounding mode: -1 Undefined 0 Round to 0 1 Round to nearest, ties to even 2 Round to ...
Definition ISDOpcodes.h:988
@ MULHU
MULHU/MULHS - Multiply high - Multiply two integers of type iN, producing an unsigned/signed value of...
Definition ISDOpcodes.h:714
@ SHL
Shift and rotation operations.
Definition ISDOpcodes.h:779
@ VECTOR_SHUFFLE
VECTOR_SHUFFLE(VEC1, VEC2) - Returns a vector, of the same type as VEC1/VEC2.
Definition ISDOpcodes.h:659
@ STRICT_FMAXIMUM
Definition ISDOpcodes.h:475
@ EXTRACT_VECTOR_ELT
EXTRACT_VECTOR_ELT(VECTOR, IDX) - Returns a single element from VECTOR identified by the (potentially...
Definition ISDOpcodes.h:581
@ ZERO_EXTEND
ZERO_EXTEND - Used for integer types, zeroing the new bits.
Definition ISDOpcodes.h:868
@ SELECT_CC
Select with condition operator - This selects between a true value and a false value (ops #2 and #3) ...
Definition ISDOpcodes.h:829
@ FMINNUM
FMINNUM/FMAXNUM - Perform floating-point minimum maximum on two values, following IEEE-754 definition...
@ DYNAMIC_STACKALLOC
DYNAMIC_STACKALLOC - Allocate some number of bytes on the stack aligned to a specified boundary.
@ ANY_EXTEND_VECTOR_INREG
ANY_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register any-extension of the low la...
Definition ISDOpcodes.h:917
@ SIGN_EXTEND_INREG
SIGN_EXTEND_INREG - This operator atomically performs a SHL/SRA pair to sign extend a small value in ...
Definition ISDOpcodes.h:906
@ SMIN
[US]{MIN/MAX} - Binary minimum or maximum of signed or unsigned integers.
Definition ISDOpcodes.h:737
@ FP_EXTEND
X = FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:996
@ VSELECT
Select with a vector condition (op #0) and two vector operands (ops #1 and #2), returning a vector re...
Definition ISDOpcodes.h:823
@ UADDO_CARRY
Carry-using nodes for multiple precision addition and subtraction.
Definition ISDOpcodes.h:331
@ STRICT_SINT_TO_FP
STRICT_[US]INT_TO_FP - Convert a signed or unsigned integer to a floating point value.
Definition ISDOpcodes.h:489
@ STRICT_FROUNDEVEN
Definition ISDOpcodes.h:469
@ FRAMEADDR
FRAMEADDR, RETURNADDR - These nodes represent llvm.frameaddress and llvm.returnaddress on the DAG.
Definition ISDOpcodes.h:112
@ STRICT_FP_TO_UINT
Definition ISDOpcodes.h:483
@ STRICT_FP_ROUND
X = STRICT_FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision ...
Definition ISDOpcodes.h:505
@ STRICT_FP_TO_SINT
STRICT_FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:482
@ FMINIMUM
FMINIMUM/FMAXIMUM - NaN-propagating minimum/maximum that also treat -0.0 as less than 0....
@ FP_TO_SINT
FP_TO_[US]INT - Convert a floating point value to a signed or unsigned integer.
Definition ISDOpcodes.h:944
@ READCYCLECOUNTER
READCYCLECOUNTER - This corresponds to the readcyclecounter intrinsic.
@ STRICT_FP_EXTEND
X = STRICT_FP_EXTEND(Y) - Extend a smaller FP type into a larger FP type.
Definition ISDOpcodes.h:510
@ AND
Bitwise operators - logical and, logical or, logical xor.
Definition ISDOpcodes.h:749
@ TRAP
TRAP - Trapping instruction.
@ INTRINSIC_WO_CHAIN
RESULT = INTRINSIC_WO_CHAIN(INTRINSICID, arg1, arg2, ...) This node represents a target intrinsic fun...
Definition ISDOpcodes.h:207
@ STRICT_FADD
Constrained versions of the binary floating point operators.
Definition ISDOpcodes.h:430
@ INSERT_VECTOR_ELT
INSERT_VECTOR_ELT(VECTOR, VAL, IDX) - Returns VECTOR with the element at IDX replaced with VAL.
Definition ISDOpcodes.h:570
@ TokenFactor
TokenFactor - This node takes multiple tokens as input and produces a single token result.
Definition ISDOpcodes.h:55
@ ATOMIC_SWAP
Val, OUTCHAIN = ATOMIC_SWAP(INCHAIN, ptr, amt) Val, OUTCHAIN = ATOMIC_LOAD_[OpName](INCHAIN,...
@ CTTZ_ZERO_POISON
Bit counting operators with a poisoned result for zero inputs.
Definition ISDOpcodes.h:805
@ FP_ROUND
X = FP_ROUND(Y, TRUNC) - Rounding 'Y' from a larger floating point type down to the precision of the ...
Definition ISDOpcodes.h:977
@ ZERO_EXTEND_VECTOR_INREG
ZERO_EXTEND_VECTOR_INREG(Vector) - This operator represents an in-register zero-extension of the low ...
Definition ISDOpcodes.h:939
@ ADDRSPACECAST
ADDRSPACECAST - This operator converts between pointers of different address spaces.
@ STRICT_FNEARBYINT
Definition ISDOpcodes.h:461
@ EH_SJLJ_SETJMP
RESULT, OUTCHAIN = EH_SJLJ_SETJMP(INCHAIN, buffer) This corresponds to the eh.sjlj....
Definition ISDOpcodes.h:164
@ TRUNCATE
TRUNCATE - Completely drop the high bits.
Definition ISDOpcodes.h:874
@ BRCOND
BRCOND - Conditional branch.
@ SHL_PARTS
SHL_PARTS/SRA_PARTS/SRL_PARTS - These operators are used for expanded integer shift operations.
Definition ISDOpcodes.h:851
@ AssertSext
AssertSext, AssertZext - These nodes record if a register contains a value that has already been zero...
Definition ISDOpcodes.h:64
@ FCOPYSIGN
FCOPYSIGN(X, Y) - Return the value of X with the sign of Y.
Definition ISDOpcodes.h:539
@ GET_DYNAMIC_AREA_OFFSET
GET_DYNAMIC_AREA_OFFSET - get offset from native SP to the address of the most recent dynamic alloca.
@ FMINIMUMNUM
FMINIMUMNUM/FMAXIMUMNUM - minimumnum/maximumnum that is same with FMINNUM_IEEE and FMAXNUM_IEEE besid...
@ INTRINSIC_W_CHAIN
RESULT,OUTCHAIN = INTRINSIC_W_CHAIN(INCHAIN, INTRINSICID, arg1, ...) This node represents a target in...
Definition ISDOpcodes.h:215
@ BUILD_VECTOR
BUILD_VECTOR(ELT0, ELT1, ELT2, ELT3,...) - Return a fixed-width vector with the specified,...
Definition ISDOpcodes.h:561
bool isNormalStore(const SDNode *N)
Returns true if the specified node is a non-truncating and unindexed store.
LLVM_ABI bool isConstantSplatVectorAllZeros(const SDNode *N, bool BuildVectorOnly=false)
Return true if the specified node is a BUILD_VECTOR or SPLAT_VECTOR where all of the elements are 0 o...
LLVM_ABI CondCode getSetCCInverse(CondCode Operation, EVT Type)
Return the operation corresponding to !(X op Y), where 'op' is a valid SetCC operation.
LLVM_ABI CondCode getSetCCSwappedOperands(CondCode Operation)
Return the operation corresponding to (Y op X) when given the operation for (X op Y).
LLVM_ABI bool isBuildVectorAllZeros(const SDNode *N)
Return true if the specified node is a BUILD_VECTOR where all of the elements are 0 or undef.
LLVM_ABI bool isConstantSplatVector(const SDNode *N, APInt &SplatValue)
Node predicates.
CondCode
ISD::CondCode enum - These are ordered carefully to make the bitfields below work out,...
LoadExtType
LoadExtType enum - This enum defines the three variants of LOADEXT (load with extension).
bool isNormalLoad(const SDNode *N)
Returns true if the specified node is a non-extending and unindexed load.
Flag
These should be considered private to the implementation of the MCInstrDesc class.
BinaryOp_match< LHS, RHS, Instruction::And > m_And(const LHS &L, const RHS &R)
auto m_Cmp()
Matches any compare instruction and ignore it.
ap_match< APInt > m_APInt(const APInt *&Res)
Match a ConstantInt or splatted ConstantVector, binding the specified pointer to the contained APInt.
bool match(Val *V, const Pattern &P)
auto m_Value()
Match an arbitrary value and ignore it.
LLVM_ABI Libcall getSINTTOFP(EVT OpVT, EVT RetVT)
getSINTTOFP - Return the SINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getUINTTOFP(EVT OpVT, EVT RetVT)
getUINTTOFP - Return the UINTTOFP_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
LLVM_ABI Libcall getFPTOSINT(EVT OpVT, EVT RetVT)
getFPTOSINT - Return the FPTOSINT_*_* value for the given types, or UNKNOWN_LIBCALL if there is none.
@ System
Synchronized with respect to all concurrently executing threads.
Definition LLVMContext.h:58
const unsigned GR64Regs[16]
const unsigned VR128Regs[32]
const unsigned VR16Regs[32]
const unsigned GR128Regs[16]
const unsigned FP32Regs[16]
const unsigned FP16Regs[16]
const unsigned GR32Regs[16]
const unsigned FP64Regs[16]
const int64_t ELFCallFrameSize
const unsigned VR64Regs[32]
const unsigned FP128Regs[16]
const unsigned VR32Regs[32]
unsigned odd128(bool Is32bit)
const unsigned CCMASK_CMP_GE
Definition SystemZ.h:42
static bool isImmHH(uint64_t Val)
Definition SystemZ.h:178
const unsigned CCMASK_TEND
Definition SystemZ.h:99
const unsigned CCMASK_CS_EQ
Definition SystemZ.h:69
const unsigned CCMASK_TBEGIN
Definition SystemZ.h:94
const unsigned CCMASK_0
Definition SystemZ.h:29
const MCPhysReg ELFArgFPRs[ELFNumArgFPRs]
MachineBasicBlock * splitBlockBefore(MachineBasicBlock::iterator MI, MachineBasicBlock *MBB)
const unsigned CCMASK_TM_SOME_1
Definition SystemZ.h:84
const unsigned CCMASK_LOGICAL_CARRY
Definition SystemZ.h:62
const unsigned TDCMASK_NORMAL_MINUS
Definition SystemZ.h:124
const unsigned CCMASK_TDC
Definition SystemZ.h:111
const unsigned CCMASK_FCMP
Definition SystemZ.h:50
const unsigned CCMASK_TM_SOME_0
Definition SystemZ.h:83
static bool isImmHL(uint64_t Val)
Definition SystemZ.h:173
const unsigned TDCMASK_SUBNORMAL_MINUS
Definition SystemZ.h:126
const unsigned PFD_READ
Definition SystemZ.h:117
const unsigned CCMASK_1
Definition SystemZ.h:30
const unsigned TDCMASK_NORMAL_PLUS
Definition SystemZ.h:123
const unsigned PFD_WRITE
Definition SystemZ.h:118
const unsigned CCMASK_CMP_GT
Definition SystemZ.h:39
const unsigned TDCMASK_QNAN_MINUS
Definition SystemZ.h:130
const unsigned CCMASK_CS
Definition SystemZ.h:71
const unsigned CCMASK_ANY
Definition SystemZ.h:33
const unsigned CCMASK_ARITH
Definition SystemZ.h:57
const unsigned CCMASK_TM_MIXED_MSB_0
Definition SystemZ.h:80
const unsigned TDCMASK_SUBNORMAL_PLUS
Definition SystemZ.h:125
static bool isImmLL(uint64_t Val)
Definition SystemZ.h:163
const unsigned VectorBits
Definition SystemZ.h:156
static bool isImmLH(uint64_t Val)
Definition SystemZ.h:168
MachineBasicBlock * emitBlockAfter(MachineBasicBlock *MBB)
const unsigned TDCMASK_INFINITY_PLUS
Definition SystemZ.h:127
unsigned reverseCCMask(unsigned CCMask)
const unsigned CCMASK_TM_ALL_0
Definition SystemZ.h:79
const unsigned IPM_CC
Definition SystemZ.h:114
const unsigned CCMASK_CMP_LE
Definition SystemZ.h:41
const unsigned CCMASK_CMP_O
Definition SystemZ.h:46
const unsigned CCMASK_CMP_EQ
Definition SystemZ.h:37
const unsigned VectorBytes
Definition SystemZ.h:160
const unsigned TDCMASK_INFINITY_MINUS
Definition SystemZ.h:128
const unsigned CCMASK_ICMP
Definition SystemZ.h:49
const unsigned CCMASK_VCMP_ALL
Definition SystemZ.h:103
const unsigned CCMASK_VCMP_NONE
Definition SystemZ.h:105
MachineBasicBlock * splitBlockAfter(MachineBasicBlock::iterator MI, MachineBasicBlock *MBB)
const unsigned CCMASK_VCMP
Definition SystemZ.h:106
const unsigned CCMASK_TM_MIXED_MSB_1
Definition SystemZ.h:81
const unsigned CCMASK_TM_MSB_0
Definition SystemZ.h:85
const unsigned CCMASK_ARITH_OVERFLOW
Definition SystemZ.h:56
const unsigned CCMASK_CS_NE
Definition SystemZ.h:70
const unsigned TDCMASK_SNAN_PLUS
Definition SystemZ.h:131
const unsigned CCMASK_TM
Definition SystemZ.h:87
const unsigned CCMASK_3
Definition SystemZ.h:32
const unsigned CCMASK_NONE
Definition SystemZ.h:28
const unsigned CCMASK_CMP_LT
Definition SystemZ.h:38
const unsigned CCMASK_CMP_NE
Definition SystemZ.h:40
const unsigned TDCMASK_ZERO_PLUS
Definition SystemZ.h:121
const unsigned TDCMASK_QNAN_PLUS
Definition SystemZ.h:129
const unsigned TDCMASK_ZERO_MINUS
Definition SystemZ.h:122
unsigned even128(bool Is32bit)
const unsigned CCMASK_TM_ALL_1
Definition SystemZ.h:82
const unsigned CCMASK_LOGICAL_BORROW
Definition SystemZ.h:64
const unsigned ELFNumArgFPRs
const unsigned CCMASK_CMP_UO
Definition SystemZ.h:45
const unsigned CCMASK_LOGICAL
Definition SystemZ.h:66
const unsigned CCMASK_TM_MSB_1
Definition SystemZ.h:86
const unsigned TDCMASK_SNAN_MINUS
Definition SystemZ.h:132
initializer< Ty > init(const Ty &Val)
support::ulittle32_t Word
Definition IRSymtab.h:53
@ User
could "use" a pointer
NodeAddr< UseNode * > Use
Definition RDFGraph.h:385
NodeAddr< NodeBase * > Node
Definition RDFGraph.h:381
NodeAddr< CodeNode * > Code
Definition RDFGraph.h:388
This is an optimization pass for GlobalISel generic memory operations.
@ Low
Lower the current thread's priority such that it does not affect foreground tasks significantly.
Definition Threading.h:280
unsigned Log2_32_Ceil(uint32_t Value)
Return the ceil log base 2 of the specified value, 32 if the value is zero.
Definition MathExtras.h:339
@ Offset
Definition DWP.cpp:577
@ Length
Definition DWP.cpp:577
MachineInstrBuilder BuildMI(MachineFunction &MF, const MIMetadata &MIMD, const MCInstrDesc &MCID)
Builder interface. Specify how to create the initial instruction itself.
constexpr bool isInt(int64_t x)
Checks if an integer fits into the given bit width.
Definition MathExtras.h:166
LLVM_ABI bool isNullConstant(SDValue V)
Returns true if V is a constant integer zero.
@ Known
Known to have no common set bits.
@ Define
Register definition.
LLVM_ABI SDValue peekThroughBitcasts(SDValue V)
Return the non-bitcasted source operand of V if it exists.
decltype(auto) dyn_cast(const From &Val)
dyn_cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:643
@ Done
Definition Threading.h:60
@ Load
The value being inserted comes from a load (InsertElement only).
testing::Matcher< const detail::ErrorHolder & > Failed()
Definition Error.h:198
iterator_range< T > make_range(T x, T y)
Convenience function for iterating over sub-ranges.
constexpr T maskLeadingOnes(unsigned N)
Create a bitmask with the N left-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:89
constexpr bool isUIntN(unsigned N, uint64_t x)
Checks if an unsigned integer fits into the given (dynamic) bit width.
Definition MathExtras.h:244
LLVM_ABI void dumpBytes(ArrayRef< uint8_t > Bytes, raw_ostream &OS)
Convert ‘Bytes’ to a hex string and output to ‘OS’.
T bit_ceil(T Value)
Returns the smallest integral power of two no smaller than Value if Value is nonzero.
Definition bit.h:362
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Value
Definition InstrProf.h:143
int countr_zero(T Val)
Count number of 0's from the least significant bit to the most stopping at the first 1.
Definition bit.h:204
int countl_zero(T Val)
Count number of 0's from the most significant bit to the least stopping at the first 1.
Definition bit.h:263
LLVM_ABI bool isBitwiseNot(SDValue V, bool AllowUndefs=false)
Returns true if V is a bitwise not operation.
constexpr bool isPowerOf2_32(uint32_t Value)
Return true if the argument is a power of two > 0.
Definition MathExtras.h:280
LLVM_ABI raw_ostream & dbgs()
dbgs() - This returns a reference to a raw_ostream for debugging messages.
Definition Debug.cpp:209
LLVM_ABI void report_fatal_error(Error Err, bool gen_crash_diag=true)
Definition Error.cpp:163
constexpr bool isUInt(uint64_t x)
Checks if an unsigned integer fits into the given bit width.
Definition MathExtras.h:190
class LLVM_GSL_OWNER SmallVector
Forward declaration of SmallVector so that calculateSmallVectorDefaultInlinedElements can reference s...
bool isa(const From &Val)
isa<X> - Return true if the parameter to the template is an instance of one of the template type argu...
Definition Casting.h:547
@ Success
The lock was released successfully.
LLVM_ABI raw_fd_ostream & errs()
This returns a reference to a raw_ostream for standard error.
AtomicOrdering
Atomic ordering for LLVM's memory model.
constexpr T divideCeil(U Numerator, V Denominator)
Returns the integer ceil(Numerator / Denominator).
Definition MathExtras.h:389
@ First
Helpers to iterate all locations in the MemoryEffectsBase class.
Definition ModRef.h:74
@ BeforeLegalizeTypes
Definition DAGCombine.h:16
uint16_t MCPhysReg
An unsigned integer type large enough to represent all physical registers, but not necessarily virtua...
Definition MCRegister.h:21
@ Fast
Assign the register banks as fast as possible (default).
RelativeUniformCounterPtr ValuesPtrExpr VTableAddr Count
Definition InstrProf.h:145
DWARFExpression::Operation Op
ArrayRef(const T &OneElt) -> ArrayRef< T >
LLVM_ABI ConstantSDNode * isConstOrConstSplat(SDValue N, bool AllowUndefs=false, bool AllowTruncation=false)
Returns the SDNode if it is a constant splat BuildVector or constant int.
constexpr unsigned BitWidth
ExceptionHandling
Definition CodeGen.h:54
decltype(auto) cast(const From &Val)
cast<X> - Return the argument parameter cast to the specified type.
Definition Casting.h:559
UndefPoisonKind
Enumeration to track whether we are interested in Undef, Poison, or both.
Definition UndefPoison.h:20
constexpr int64_t SignExtend64(uint64_t x)
Sign-extend the number in the bottom B bits of X to a 64-bit integer.
Definition MathExtras.h:567
constexpr T maskTrailingOnes(unsigned N)
Create a bitmask with the N right-most bits set to 1, and all other bits set to 0.
Definition MathExtras.h:78
T bit_floor(T Value)
Returns the largest integral power of two no greater than Value if Value is nonzero.
Definition bit.h:347
LLVM_ABI bool isAllOnesConstant(SDValue V)
Returns true if V is an integer constant with all bits set.
MCRegisterClass TargetRegisterClass
Definition FastISel.h:58
void swap(llvm::BitVector &LHS, llvm::BitVector &RHS)
Implement std::swap in terms of BitVector swap.
Definition BitVector.h:880
#define N
#define EQ(a, b)
Definition regexec.c:65
AddressingMode(bool LongDispl, bool IdxReg)
This struct is a compact representation of a valid (non-zero power of two) alignment.
Definition Alignment.h:39
Extended Value Type.
Definition ValueTypes.h:35
EVT changeVectorElementTypeToInteger() const
Return a vector with the same number of elements as this vector, but with the element type converted ...
Definition ValueTypes.h:90
TypeSize getStoreSize() const
Return the number of bytes overwritten by a store of the specified value type.
Definition ValueTypes.h:418
bool isSimple() const
Test if the given EVT is simple (as opposed to being extended).
Definition ValueTypes.h:145
static EVT getVectorVT(LLVMContext &Context, EVT VT, unsigned NumElements, bool IsScalable=false)
Returns the EVT that represents a vector NumElements in length, where each element is of type VT.
Definition ValueTypes.h:70
bool isFloatingPoint() const
Return true if this is a FP or a vector FP type.
Definition ValueTypes.h:155
TypeSize getSizeInBits() const
Return the size of the specified value type in bits.
Definition ValueTypes.h:396
uint64_t getScalarSizeInBits() const
Definition ValueTypes.h:408
MVT getSimpleVT() const
Return the SimpleValueType held in the specified simple EVT.
Definition ValueTypes.h:339
static EVT getIntegerVT(LLVMContext &Context, unsigned BitWidth)
Returns the EVT that represents an integer with the given number of bits.
Definition ValueTypes.h:61
uint64_t getFixedSizeInBits() const
Return the size of the specified fixed width value type in bits.
Definition ValueTypes.h:404
bool isVector() const
Return true if this is a vector value type.
Definition ValueTypes.h:176
EVT getScalarType() const
If this is a vector type, return the element type, otherwise return this.
Definition ValueTypes.h:346
LLVM_ABI Type * getTypeForEVT(LLVMContext &Context) const
This method returns an LLVM type corresponding to the specified EVT.
bool isRound() const
Return true if the size is a power-of-two number of bytes.
Definition ValueTypes.h:271
EVT getVectorElementType() const
Given a vector type, return the type of each element.
Definition ValueTypes.h:351
bool isVectorOf(EVT EltVT) const
Return true if this is a vector with matching element type.
Definition ValueTypes.h:181
bool isScalarInteger() const
Return true if this is an integer, but not a vector.
Definition ValueTypes.h:165
unsigned getVectorNumElements() const
Given a vector type, return the number of elements it contains.
Definition ValueTypes.h:359
bool isInteger() const
Return true if this is an integer or a vector integer type.
Definition ValueTypes.h:160
KnownBits intersectWith(const KnownBits &RHS) const
Returns KnownBits information that is known to be true for both this and RHS.
Definition KnownBits.h:325
APInt getMaxValue() const
Return the maximal unsigned value possible given these KnownBits.
Definition KnownBits.h:146
This class contains a discriminated union of information about pointers in memory operands,...
static LLVM_ABI MachinePointerInfo getConstantPool(MachineFunction &MF)
Return a MachinePointerInfo record that refers to the constant pool.
MachinePointerInfo getWithOffset(int64_t O) const
static LLVM_ABI MachinePointerInfo getGOT(MachineFunction &MF)
Return a MachinePointerInfo record that refers to a GOT entry.
static LLVM_ABI MachinePointerInfo getFixedStack(MachineFunction &MF, int FI, int64_t Offset=0)
Return a MachinePointerInfo record that refers to the specified FrameIndex.
This represents a list of ValueType's that has been intern'd by a SelectionDAG.
SmallVector< unsigned, 2 > OpVals
bool isVectorConstantLegal(const SystemZSubtarget &Subtarget)
This represents an addressing mode of: BaseGV + BaseOffs + BaseReg + Scale*ScaleReg + ScalableOffset*...
This contains information for each constraint that we are lowering.
This structure contains all information that is necessary for lowering calls.
SmallVector< ISD::InputArg, 32 > Ins
CallLoweringInfo & setDiscardResult(bool Value=true)
CallLoweringInfo & setZExtResult(bool Value=true)
CallLoweringInfo & setDebugLoc(const SDLoc &dl)
CallLoweringInfo & setSExtResult(bool Value=true)
CallLoweringInfo & setNoReturn(bool Value=true)
SmallVector< ISD::OutputArg, 32 > Outs
CallLoweringInfo & setChain(SDValue InChain)
CallLoweringInfo & setCallee(CallingConv::ID CC, Type *ResultType, SDValue Target, ArgListTy &&ArgsList, AttributeSet ResultAttrs={})
This structure is used to pass arguments to makeLibCall function.